我正在计算平均时间以得到关于Stack Overflow的答复,结果没有任何意义。
#standardSQL
WITH question_answers AS (
SELECT *
, timestamp_diff(answers.first, creation_date, minute) minutes
FROM (
SELECT creation_date
, (SELECT AS STRUCT MIN(creation_date) first, COUNT(*) c
FROM `bigquery-public-data.stackoverflow.posts_answers` b
WHERE a.id=b.parent_id
) answers
, SPLIT(tags, '|') tags
FROM `bigquery-public-data.stackoverflow.posts_questions` a
WHERE EXTRACT(year FROM creation_date) > 2015
), UNNEST(tags) tag
WHERE tag IN ('java', 'javascript', 'google-bigquery', 'firebase', 'php')
AND answers.c > 0
)
SELECT tag
, COUNT(*) questions
, ROUND(AVG(minutes), 2) first_reply_avg_minutes
FROM question_answers
GROUP BY tag
我应该如何计算平均时间?
答案 0 :(得分:2)
实际上-在平均超过100小时(> 6000分钟)的堆栈溢出中获得答案似乎是错误的-很大程度上是由异常值驱动的。
代替简单的AVG()
,您可以获得:
EXP(AVG(LOG(GREATEST(minutes,1))))
AVG(q) FROM (SELECT q FROM QUANTILES(q, 100) LIMIT 80 OFFSET 2))
。all_minutes[OFFSET(CAST(ARRAY_LENGTH(all_minutes)/2 AS INT64))]
如果使用以下任何一种方法,结果将更有意义:
正如您在此处看到的,在这种情况下,除去异常值可以得出类似于几何平均值的结果-而中位数报告的数字甚至更低。使用哪一个?您的选择。
WITH question_answers AS (
SELECT *
, timestamp_diff(answers.first, creation_date, minute) minutes
FROM (
SELECT creation_date
, (SELECT AS STRUCT MIN(creation_date) first, COUNT(*) c
FROM `bigquery-public-data.stackoverflow.posts_answers` b
WHERE a.id=b.parent_id
) answers
, SPLIT(tags, '|') tags
FROM `bigquery-public-data.stackoverflow.posts_questions` a
WHERE EXTRACT(year FROM creation_date) > 2015
), UNNEST(tags) tag
WHERE tag IN ('java', 'javascript', 'google-bigquery', 'firebase', 'php', 'sql', 'elasticsearch', 'apache-kafka', 'tensorflow')
AND answers.c > 0
)
SELECT * EXCEPT(qs, all_minutes)
, (SELECT ROUND(AVG(q),2) FROM (SELECT q FROM UNNEST(qs) q ORDER BY q LIMIT 80 OFFSET 2)) avg_no_outliers
, all_minutes[OFFSET(CAST(ARRAY_LENGTH(all_minutes)/2 AS INT64) )] median_minutes
FROM (
SELECT tag
, COUNT(*) questions
, ROUND(AVG(minutes), 2) avg_minutes
, ROUND(EXP(AVG(LOG(GREATEST(minutes,1)))),2) first_reply_avg_minutes_geom
, APPROX_QUANTILES(minutes, 100) qs
, ARRAY_AGG(minutes IGNORE NULLS ORDER BY minutes) all_minutes
FROM question_answers
GROUP BY tag
)
ORDER BY 2 DESC
奖金MEDIAN()
UDF function from Elliott。
CREATE TEMP FUNCTION MEDIAN(arr ANY TYPE) AS ((
SELECT
IF(
MOD(ARRAY_LENGTH(arr), 2) = 0,
(arr[OFFSET(DIV(ARRAY_LENGTH(arr), 2) - 1)] + arr[OFFSET(DIV(ARRAY_LENGTH(arr), 2))]) / 2,
arr[OFFSET(DIV(ARRAY_LENGTH(arr), 2))]
)
FROM (SELECT ARRAY_AGG(x ORDER BY x) AS arr FROM UNNEST(arr) AS x)
));