arrays - 从 R 中的另一个 3D 数组填充 3D 数组的最快方法-6ren

arrays - 从 R 中的另一个 3D 数组填充 3D 数组的最快方法

转载作者：行者123 更新时间：2023-12-01 03:10:53

24

4

我正在使用下面的代码从另一个 3D 数组填充 3D 数组。我用过 sapply函数在每个人(第三维)应用代码行，如 Efficient way to fill a 3D array .
这是我的代码。

ind <- 1000
    individuals <- as.character(seq(1, ind, by = 1))
    maxCol <- 7
    col <- 4
    line <- 0
    a <- 0
    b <- 0
    c <- 0

    col_array <- c("year","time", "ID", "age", as.vector(outer(c(paste(seq(0, 1, by = 1), "year", sep="_"), paste(seq(2, maxCol, by = 1), "years", sep="_")), c("S_F", "I_F", "R_F"), paste, sep="_")))
    array1 <- array(sample(1:100, length(col_array), replace = T), dim=c(2, length(col_array), ind), dimnames=list(NULL, col_array, individuals)) ## 3rd dimension = individual ID
    ## print(array1)

    col_array <- c("year","time", "ID", "age", as.vector(outer(c(paste(seq(0, 1, by = 1), "year", sep="_"), paste(seq(2, maxCol, by = 1), "years", sep="_")), c("S_M", "I_M", "R_M"), paste, sep="_")))
    array2 <- array(NA, dim=c(2, length(col_array), ind), dimnames=list(NULL, col_array, individuals)) ## 3rd dimension = individual ID
    ## print(array2)

    tic("array2")
    array2 <- sapply(individuals, function(i){

      ## Fill the first columns
      array2[line + 1, c("year", "time", "ID", "age"), i] <- c(a, b, i, c)

      ## Define column indexes for individuals S
      col_start_S_F <- which(colnames(array1[,,i])=="0_year_S_F")
      col_end_S_F <- which(colnames(array1[,,i])==paste(maxCol,"years_S_F", sep="_"))
      col_start_S_M <- which(colnames(array2[,,i])=="0_year_S_M")
      col_end_S_M <- which(colnames(array2[,,i])==paste(maxCol,"years_S_M", sep="_"))

      ## Fill the columns for individuals S
      p_S_M <- sapply(0:maxCol, function(x){pnorm(x, 4, 1)})
      array2[line + 1, col_start_S_M:col_end_S_M, i] <- round(as.numeric(as.vector(array1[line + 1, col_start_S_F:col_end_S_F, i]))*p_S_M)

      ## Define column indexes for individuals I
      col_start_I_F <- which(colnames(array1[,,i])=="0_year_I_F")
      col_end_I_F <- which(colnames(array1[,,i])==paste(maxCol,"years_I_F", sep="_"))
      col_start_I_M <- which(colnames(array2[,,i])=="0_year_I_M")
      col_end_I_M <- which(colnames(array2[,,i])==paste(maxCol,"years_I_M", sep="_"))

      ## Fill the columns for individuals I
      p_I_M <- sapply(0:maxCol, function(x){pnorm(x, 2, 1)})
      array2[line + 1, col_start_I_M:col_end_I_M, i] <- round(as.numeric(as.vector(array1[line + 1, col_start_I_F:col_end_I_F, i]))*p_I_M)

      ## Define column indexes for individuals R
      col_start_R_M <- which(colnames(array2[,,i])=="0_year_R_M")
      col_end_R_M <- which(colnames(array2[,,i])==paste(maxCol,"years_R_M", sep="_"))

      ## Fill the columns for individuals R
      array2[line + 1, col_start_R_M:col_end_R_M, i] <- as.numeric(as.vector(array2[line + 1, col_start_S_M:col_end_S_M, i])) + 
        as.numeric(as.vector(array2[line + 1, col_start_I_M:col_end_I_M, i]))

      return(array2[,,i])
      ## print(array2[,,i])

    }, simplify = "array") 
    ## print(array2)
    toc()

有没有办法提高我的代码的性能/速度(即 < 1 秒)？第 3 维有 500000 个观测值。有什么建议？

最佳答案

TL;DR:这是一个 tidyverse 解决方案，它将样本数组转换为数据帧并应用请求的更改。 编辑:我添加了步骤 1+2 将原始帖子的示例数据转换为我在步骤 3 中使用的格式。步骤 3 中的实际计算非常快(<0.1 秒)，但瓶颈是步骤 2，它需要50 万行需要 10 秒。

步骤 0:为 50 万人创建样本数据

ind <- 500000
individuals <- as.character(seq(1, ind, by = 1))
maxCol <- 7
col <- 4
line <- 0
a <- 0
b <- 0
c <- 0

col_array <- c("year","time", "ID", "age", as.vector(outer(c(paste(seq(0, 1, by = 1), "year", sep="_"), paste(seq(2, maxCol, by = 1), "years", sep="_")), c("S_F", "I_F", "R_F"), paste, sep="_")))
array1 <- array(sample(1:100, length(col_array), replace = T), dim=c(2, length(col_array), ind), dimnames=list(NULL, col_array, individuals)) ## 3rd dimension = individual ID

dim(array1)
# [1]      2     28 500000    # Two rows x 28 measures x 500k individuals

步骤 1:子集数组并转换为数据帧。

library(tidyverse)
# OP only uses first line of array1. If other rows needed, replace with "array1 %>%"
#   and adjust renaming below to account for different Var1.
array1_dt <- array1[1,,] %>% 
  as.data.frame.table(stringsAsFactors = FALSE)

第 2 步:将统计数据分成不同的列，每个年份占一行。这是最慢的一步(尤其是 spread 行)，1000 个人需要 0.05 秒，500k 需要 10 秒。如果需要，我希望 data.table 解决方案可以使它更快。

array1_dt_reshape <- array1_dt %>%
  rename(stat = Var1, ID = Var2) %>%
  filter(!stat %in% c("year", "time", "ID", "age")) %>%
  mutate(year = stat %>% str_sub(end = 1),
         col  = stat %>% str_sub(start = -3)) %>%
  select(-stat) %>%
  spread(col, Freq) %>%
  arrange(ID)

第 3 步:应用请求的转换。此函数使用两组参数计算分布，并使用这些参数来缩放输入表的列。 500k 个人需要 0.03 秒。

array_transform <- function(input_data = array1_dt_reshape, 
                           max_yr = 7, S_M_mean = 4, I_M_mean = 2) {
  tictoc::tic()
  # First calculate the distribution function values to apply to all individuals, 
  #   depending on year.
  p_S_M_vals <- sapply(0:max_yr, function(x){pnorm(x, S_M_mean, 1)})
  p_I_M_vals <- sapply(0:max_yr, function(x){pnorm(x, I_M_mean, 1)})

  # For each year, scale S_M + I_M by the respective distribution functions.
  #   This solution relies on the fact that each ID has 8 rows every time,
  #   so we can recycle the 8 values in the distribution functions.
  output <- input_data %>% 
    # group_by(ID) %>%  <-- Not needed
    mutate(S_M = S_F * p_S_M_vals,
           I_M = I_F * p_I_M_vals,
           R_M = S_M + I_M)  # %>% ungroup  <-- Not needed
  tictoc::toc()
  return(output)
}


array1_output <- array_transform(array1_dt_reshape)

结果

head(array1_output)
   ID year I_F R_F S_F          S_M        I_M         R_M
1   1    0  16  76  23 7.284386e-04  0.3640021   0.3647305
2   1    1  46  96  80 1.079918e-01  7.2981417   7.4061335
3   1    2  27  57  76 1.729010e+00 13.5000000  15.2290100
4   1    3  42  64  96 1.523090e+01 35.3364793  50.5673837
5   1    4  74  44  57 2.850000e+01 72.3164902 100.8164902
6   1    5  89  90  64 5.384606e+01 88.8798591 142.7259228
7   1    6  23  16  44 4.299899e+01 22.9992716  65.9982658
8   1    7  80  46  90 8.987851e+01 79.9999771 169.8784862
9   2    0  16  76  23 7.284386e-04  0.3640021   0.3647305
10  2    1  46  96  80 1.079918e-01  7.2981417   7.406133

关于arrays - 从 R 中的另一个 3D 数组填充 3D 数组的最快方法，我们在Stack Overflow上找到一个类似的问题： https://stackoverflow.com/questions/52341245/

24

4

0

文章推荐： javascript - 使用 Angular-ui-router 解决 404 问题

文章推荐： Jenkins stash 不会隐藏所有文件和文件夹

文章推荐： Python3 从以数字开头的目录导入文件

文章推荐： jquery - 将两个脚本合并为一个

arrays - 为什么是 &array != &array[0]？
在 C 中: int a[10]; printf("%p\n", a); printf("%p\n", &a[0]); 产量: 0x7fff5606c600 0x7fff5606c600 这是我所期望
arrays - 替换 Array of Array 中元素的位置
我一直在尝试运行此循环来更改基于数组的元素的位置，但出现以下错误。不太确定哪里出了问题。任何想法或想法!谢谢。 var population = [[98, 8, 45, 34, 56], [9, 1
arrays - Ruby Array of Arrays 按值分组和计数
我正在尝试获取一个 Ruby 数组数组并将其分组以计算其值。数组有一个月份和一个 bool 值: array = [["June", false], ["June", false], ["June"
javascript - array.split ("stop here") array to array of arrays in javascript 数组
所以我们的目标是在遇到某个元素时将数组分割成子数组下面的示例 array.split("stop here") ["haii", "keep", "these in the same array bu
java - Arrays.stream(array) 与 Arrays.asList(array).stream()
在this问题已经回答了两个表达式是相等的，但在这种情况下它们会产生不同的结果。对于给定的 int[] 分数，为什么会这样: Arrays.stream(scores) .forEac
arrays - 我如何制作 Perl "array of arrays of hashes"？
我认为我需要的是哈希数组的数组，但我不知道如何制作它。 Perl 能做到吗？如果是这样，代码会是什么样子？最佳答案 perldoc perldsc是了解 Perl 数据结构的好文档。关于arra
android - GSON 解析 Array in array in array
我遇到了这个问题，从 API 中我得到一个扩展 JSON，其中包含一个名为坐标的对象，该对象是一个包含数组 o 数组的数组。为了更清楚地看这个例子: "coordinates": [
arrays - Postgres JSONB : where clause for arrays of arrays
postgres 中有(v 9.5，如果重要的话): create table json_test( id varchar NOT NULL, data jsonb NOT NULL, PRIM
arrays - bash中echo "${array[@]}"echo "${array[*]}"有什么区别
我用 echo "${array[@]}" 和 echo "${array[*]}" 得到了相同的结果。如果我这样做: mkdir 假音乐； touch fakemusic/{Beatles,Sto
arrays - Array of arrays of typealias 表达式类型不明确，没有更多上下文
我正在尝试创建 typealias 对象的数组数组 - 但我收到“表达式类型不明确，没有更多上下文”编译错误。这是我的代码: typealias TestClosure = ((message: St
python - array.array 与 numpy.array
如果您在 Python 中创建一维数组，使用 NumPy 包有什么好处吗？最佳答案这完全取决于您打算如何处理数组。如果您所做的只是创建简单数据类型的数组并进行 I/O，array模块就可以了。另
arrays - Perl6 : Pushing an array to an array of arrays with one element seems not to work as expected
当我将数组推送到只有一个数组作为其唯一元素的数组数组时，为什么会得到这种数据结构？ use v6; my @d = ( [ 1 .. 3 ] ); @d.push( [ 4 .. 6 ] ); @d.
arrays - 如何从 Array{Array{Int64,2},1} 转换为 Array{Int64,2}
在 Julia 中，我想将定义为二维数组向量的数据转换为二维矩阵数组。如下例所述，我想把数据s转换成数据t，但是至今没有成功。我该如何处理这个案子？ julia> s = [[1 2 3], [4
c - 1[&array] 是 &array[sizeof(array)/sizeof(array[0])] 的合适替代品还是太混淆了？
C 没有elementsof 关键字来获取数组的元素数。所以这通常由计算 sizeof(Array)/sizeof(Array[0]) 代替但这需要重复数组变量名。1[&Array] 是指向数组后第一
arrays - 为什么我可以调用 array.some() 而不是 array.every() 联合数组类型？
所以，假设我有一个像这样的(愚蠢的)函数: function doSomething(input: number|string): boolean { if (input === 42 || in
arrays - 需要的公式 : Sort array to array -"zig-zag"
我有以下数组: a = [1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16] 我将它用于一些像这样的视觉内容: 1 2 3 4 5 6 7 8 9 10
arrays - Scala:array.toList 与 array.to[List]
我想知道数组中的 .toList 与 .to[List] 之间有什么区别。我在spark-shell中做了这个测试，结果没有区别，但我不知道用什么更好。任何意见？ scala> val l = Arr
arrays - Array.Find和IndexOf用于完全相同对象的多个元素
我很难获得完全相同对象的多个元素的当前元素索引: $b = "A","D","B","D","C","E","D","F" $b | ? { $_ -contains "D" } 替代版本: $b =
arrays - Vuetify : How to do a v-select search in array of arrays
我正在尝试使用来自我的 API 的 v-select 执行 options，我将数据放在数组数组中。 Array which I got from API 它应该是一个带有搜索的 select，因为它
arrays - char *array 和 char array[] 之间的内存区别是什么？
这个问题在这里已经有了答案: String literals: pointer vs. char array (1 个回答) 4 个月前关闭。当我执行下一个代码时 int main() {

首页

博学

6Ren·AI

商城

arrays - 从 R 中的另一个 3D 数组填充 3D 数组的最快方法