在 R 的小平面包裹中绘制每组的平均数据(显示 geom_smooth)
Plot mean data for each group in facet wraps in R (show geom_smooth)
我有“日期”、“小时”、“天”、“工作日”、“值”的时间序列数据。
我想以一种方式对数据进行分组,它为我提供了每个工作日(星期一、星期二等)的平均值图,但以一种计算特定日期平均值的方式。例如在星期一的情节中,均值应该是数据测试中所有星期一的平均值。
数据:
structure(list(Date = structure(c(1482087600, 1482084000, 1482080400,
1482076800, 1482073200, 1482069600, 1482066000, 1482062400, 1482058800,
1482055200, 1482051600, 1482048000, 1482044400, 1482040800, 1482037200,
1482033600, 1482030000, 1482026400, 1482022800, 1482019200, 1482015600,
1482012000, 1482008400, 1482004800, 1482001200, 1481997600, 1481994000,
1481990400, 1481986800, 1481983200, 1481979600, 1481976000, 1481972400,
1481968800, 1481965200, 1481961600, 1481958000, 1481954400, 1481950800,
1481947200, 1481943600, 1481940000, 1481936400, 1481932800, 1481929200,
1481925600, 1481922000, 1481918400), class = c("POSIXct", "POSIXt"
), tzone = ""), hour = c(23L, 22L, 21L, 20L, 19L, 18L, 17L, 16L,
15L, 14L, 13L, 12L, 11L, 10L, 9L, 8L, 7L, 6L, 5L, 4L, 3L, 2L,
1L, 0L, 23L, 22L, 21L, 20L, 19L, 18L, 17L, 16L, 15L, 14L, 13L,
12L, 11L, 10L, 9L, 8L, 7L, 6L, 5L, 4L, 3L, 2L, 1L, 0L), day = c(18L,
18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L,
18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 17L, 17L, 17L,
17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L,
17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L), week = c(51, 51, 51,
51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51,
51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51,
51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51), weekdays = c("Sunday",
"Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday",
"Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday",
"Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday",
"Sunday", "Sunday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday"), Online_h = c(18L, 20L, 25L, 29L, 31L,
32L, 30L, 23L, 24L, 17L, 15L, 15L, 10L, 9L, 7L, 7L, 9L, 12L,
16L, 16L, 23L, 25L, 25L, 35L, 38L, 44L, 39L, 32L, 28L, 30L, 23L,
22L, 21L, 14L, 13L, 15L, 12L, 6L, 7L, 6L, 7L, 7L, 11L, 14L, 21L,
27L, 29L, 34L)), row.names = c(NA, 48L), class = "data.frame")
我当前的代码如下所示:
df%>%
group_by(day) %>%
group_by(hour) %>%
mutate(avg_hour = mean(Value)) %>%
ggplot(aes(x=hour, y=avg_hour)) +
geom_line() +
ylab("Available drivers") +
xlab("Hours") +
facet_wrap(vars(weekdays))
生成此图的结果。
但是,所有天的平均线似乎都是一样的,但如果是每组天计算的话,应该是不同的。谁能帮我正确找到每个组的方法并将其显示在地块上?
提前谢谢你。
你的 group_by
电话不应该那样分开。
编辑:我注意到你在数据集中每小时只有一个小时,所以不清楚你想要找到...的平均值...
library(tidyverse)
df %>%
group_by(weekdays, hour) %>%
mutate(avg_drivers_online_per_hour = mean(Online_h)) %>%
group_by(weekdays) %>%
mutate(avg_drivers_online_per_weekday = mean(Online_h)) %>%
ggplot() +
geom_line(aes(x=hour, y=avg_drivers_online_per_hour)) +
geom_segment(aes(x = 0, xend = 24, y = avg_drivers_online_per_weekday, yend = avg_drivers_online_per_weekday), color = "dodgerblue2") +
ylab("Available drivers") +
xlab("Hours") +
facet_wrap(vars(weekdays))
由 reprex package (v2.0.1)
于 2021-11-08 创建
我有“日期”、“小时”、“天”、“工作日”、“值”的时间序列数据。 我想以一种方式对数据进行分组,它为我提供了每个工作日(星期一、星期二等)的平均值图,但以一种计算特定日期平均值的方式。例如在星期一的情节中,均值应该是数据测试中所有星期一的平均值。
数据:
structure(list(Date = structure(c(1482087600, 1482084000, 1482080400,
1482076800, 1482073200, 1482069600, 1482066000, 1482062400, 1482058800,
1482055200, 1482051600, 1482048000, 1482044400, 1482040800, 1482037200,
1482033600, 1482030000, 1482026400, 1482022800, 1482019200, 1482015600,
1482012000, 1482008400, 1482004800, 1482001200, 1481997600, 1481994000,
1481990400, 1481986800, 1481983200, 1481979600, 1481976000, 1481972400,
1481968800, 1481965200, 1481961600, 1481958000, 1481954400, 1481950800,
1481947200, 1481943600, 1481940000, 1481936400, 1481932800, 1481929200,
1481925600, 1481922000, 1481918400), class = c("POSIXct", "POSIXt"
), tzone = ""), hour = c(23L, 22L, 21L, 20L, 19L, 18L, 17L, 16L,
15L, 14L, 13L, 12L, 11L, 10L, 9L, 8L, 7L, 6L, 5L, 4L, 3L, 2L,
1L, 0L, 23L, 22L, 21L, 20L, 19L, 18L, 17L, 16L, 15L, 14L, 13L,
12L, 11L, 10L, 9L, 8L, 7L, 6L, 5L, 4L, 3L, 2L, 1L, 0L), day = c(18L,
18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L,
18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 18L, 17L, 17L, 17L,
17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L,
17L, 17L, 17L, 17L, 17L, 17L, 17L, 17L), week = c(51, 51, 51,
51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51,
51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51,
51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51, 51), weekdays = c("Sunday",
"Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday",
"Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday",
"Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday", "Sunday",
"Sunday", "Sunday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday", "Saturday", "Saturday", "Saturday", "Saturday",
"Saturday", "Saturday"), Online_h = c(18L, 20L, 25L, 29L, 31L,
32L, 30L, 23L, 24L, 17L, 15L, 15L, 10L, 9L, 7L, 7L, 9L, 12L,
16L, 16L, 23L, 25L, 25L, 35L, 38L, 44L, 39L, 32L, 28L, 30L, 23L,
22L, 21L, 14L, 13L, 15L, 12L, 6L, 7L, 6L, 7L, 7L, 11L, 14L, 21L,
27L, 29L, 34L)), row.names = c(NA, 48L), class = "data.frame")
我当前的代码如下所示:
df%>%
group_by(day) %>%
group_by(hour) %>%
mutate(avg_hour = mean(Value)) %>%
ggplot(aes(x=hour, y=avg_hour)) +
geom_line() +
ylab("Available drivers") +
xlab("Hours") +
facet_wrap(vars(weekdays))
生成此图的结果。
但是,所有天的平均线似乎都是一样的,但如果是每组天计算的话,应该是不同的。谁能帮我正确找到每个组的方法并将其显示在地块上? 提前谢谢你。
你的 group_by
电话不应该那样分开。
编辑:我注意到你在数据集中每小时只有一个小时,所以不清楚你想要找到...的平均值...
library(tidyverse)
df %>%
group_by(weekdays, hour) %>%
mutate(avg_drivers_online_per_hour = mean(Online_h)) %>%
group_by(weekdays) %>%
mutate(avg_drivers_online_per_weekday = mean(Online_h)) %>%
ggplot() +
geom_line(aes(x=hour, y=avg_drivers_online_per_hour)) +
geom_segment(aes(x = 0, xend = 24, y = avg_drivers_online_per_weekday, yend = avg_drivers_online_per_weekday), color = "dodgerblue2") +
ylab("Available drivers") +
xlab("Hours") +
facet_wrap(vars(weekdays))
由 reprex package (v2.0.1)
于 2021-11-08 创建