/*
* The statistics page, which counts commits per author across a window of
* recent weeks, months, quarters or years and breaks the tip of the branch
* down by the language its files are written in. The window sizes live in one
* table here, which cgit.c also resolves the max-stats setting against so a
* repository can refuse the coarser windows. Neither half of the page reads a
* diff or the contents of a blob, so the whole thing costs about one rev-list
* plus one tree read.
*/
#define USE_THE_REPOSITORY_VARIABLE
#include "cgit.h"
#include "html.h"
#include "parsing.h"
#include "shared.h"
#include "ui-shared.h"
#include "ui-stats.h"
#define DEFAULT_AUTHOR_ROWS 10
#define MAX_LANGUAGE_ROWS 6
static const struct {
const char *ext;
const char *label;
} lang_map[] = {
{"c", "C"}, {"h", "C"},
{"cpp", "C++"}, {"cc", "C++"}, {"cxx", "C++"},
{"hpp", "C++"}, {"hh", "C++"},
{"js", "JavaScript"}, {"mjs", "JavaScript"},
{"ts", "TypeScript"}, {"tsx", "TypeScript"},
{"py", "Python"}, {"lua", "Lua"},
{"sh", "Shell"}, {"bash", "Shell"},
{"go", "Go"}, {"rs", "Rust"}, {"zig", "Zig"},
{"java", "Java"}, {"kt", "Kotlin"}, {"cs", "C#"}, {"swift", "Swift"},
{"rb", "Ruby"}, {"pl", "Perl"}, {"pm", "Perl"}, {"php", "PHP"},
{"hs", "Haskell"}, {"el", "Lisp"}, {"ml", "OCaml"},
{"css", "CSS"}, {"scss", "CSS"},
{"html", "HTML"}, {"htm", "HTML"}, {"xml", "XML"},
{"md", "Markdown"}, {"rst", "Text"}, {"txt", "Text"},
{"json", "JSON"}, {"yml", "YAML"}, {"yaml", "YAML"}, {"toml", "TOML"},
{"mk", "Make"}, {"tex", "TeX"}, {"sql", "SQL"}, {"vim", "Vimscript"},
};
/*
* One author's share of the window. periods is keyed by the label of a
* period, with the commit count stored in the util field itself rather than
* behind another allocation.
*/
struct authorstat {
long total;
struct string_list periods;
};
/*
* Bytes at the tip of the branch per language. langs is keyed by the language
* name, again with the count living in the util field.
*/
struct lang_sizes {
struct string_list langs;
unsigned long total;
};
static void trunc_week(struct tm *tm)
{
time_t t = timegm(tm);
// tm_wday counts from Sunday, while the label comes from %V and %G,
// which number the ISO weeks that start on Monday.
t -= ((tm->tm_wday + 6) % 7) * SECONDS_PER_DAY;
gmtime_r(&t, tm);
}
static void dec_week(struct tm *tm)
{
time_t t = timegm(tm);
t -= SECONDS_PER_WEEK;
gmtime_r(&t, tm);
}
static void inc_week(struct tm *tm)
{
time_t t = timegm(tm);
t += SECONDS_PER_WEEK;
gmtime_r(&t, tm);
}
static char *pretty_week(struct tm *tm)
{
static char buf[10];
// A year of five digits or more does not fit, and strftime then leaves
// the buffer with contents the standard says nothing about, so empty it
// rather than return whatever the last week left.
if (!strftime(buf, sizeof(buf), "W%V %G", tm))
buf[0] = '\0';
return buf;
}
static void trunc_month(struct tm *tm)
{
tm->tm_mday = 1;
}
static void dec_month(struct tm *tm)
{
tm->tm_mon--;
if (tm->tm_mon < 0) {
tm->tm_year--;
tm->tm_mon = 11;
}
}
static void inc_month(struct tm *tm)
{
tm->tm_mon++;
if (tm->tm_mon > 11) {
tm->tm_year++;
tm->tm_mon = 0;
}
}
static char *pretty_month(struct tm *tm)
{
static const char *months[] = {
"Jan", "Feb", "Mar", "Apr", "May", "Jun",
"Jul", "Aug", "Sep", "Oct", "Nov", "Dec"
};
return cgit_fmt("%s %d", months[tm->tm_mon], tm->tm_year + 1900);
}
static void trunc_quarter(struct tm *tm)
{
trunc_month(tm);
while (tm->tm_mon % 3 != 0)
dec_month(tm);
}
static void dec_quarter(struct tm *tm)
{
dec_month(tm);
dec_month(tm);
dec_month(tm);
}
static void inc_quarter(struct tm *tm)
{
inc_month(tm);
inc_month(tm);
inc_month(tm);
}
static char *pretty_quarter(struct tm *tm)
{
return cgit_fmt("Q%d %d", tm->tm_mon / 3 + 1, tm->tm_year + 1900);
}
static void trunc_year(struct tm *tm)
{
trunc_month(tm);
tm->tm_mon = 0;
}
static void dec_year(struct tm *tm)
{
tm->tm_year--;
}
static void inc_year(struct tm *tm)
{
tm->tm_year++;
}
static char *pretty_year(struct tm *tm)
{
return cgit_fmt("%d", tm->tm_year + 1900);
}
/*
* The order runs from the finest window to the coarsest, because a repository
* caps the page by storing an index into this table as its max-stats.
*/
static const struct cgit_period periods[] = {
{'w', "week", 12, 4, trunc_week, dec_week, inc_week, pretty_week},
{'m', "month", 12, 4, trunc_month, dec_month, inc_month, pretty_month},
{'q', "quarter", 12, 4, trunc_quarter, dec_quarter, inc_quarter, pretty_quarter},
{'y', "year", 12, 4, trunc_year, dec_year, inc_year, pretty_year},
};
static void window_start(const struct cgit_period *period, struct tm *tm)
{
time_t now;
int i;
time(&now);
gmtime_r(&now, tm);
period->trunc(tm);
for (i = 1; i < period->count; i++)
period->dec(tm);
}
static void add_commit(struct string_list *authors, struct commitinfo *info,
const struct cgit_period *period)
{
struct string_list_item *author, *bucket;
struct authorstat *stats;
char *name, *label;
struct tm date;
time_t when;
uintptr_t *count;
// A commit can lack an author header, so fall back rather than
// handing xstrdup a NULL.
name = xstrdup(info->author ? info->author : "(unknown)");
author = string_list_insert(authors, name);
if (!author->util)
author->util = xcalloc(1, sizeof(struct authorstat));
else
free(name);
stats = author->util;
when = info->committer_date;
// A crafted commit can carry a date outside the range gmtime_r can
// represent, which would leave date uninitialized and later index the
// month table out of bounds.
if (!gmtime_r(&when, &date))
return;
period->trunc(&date);
label = xstrdup(period->pretty(&date));
bucket = string_list_insert(&stats->periods, label);
count = (uintptr_t *)&bucket->util;
if (*count)
free(label);
(*count)++;
stats->total++;
}
/*
* Count the commits in the displayed window, returning a list of authors
* whose util field holds a struct authorstat. Merge commits are left out, so
* that pulling a branch in does not credit the merger with its commits.
*/
static struct string_list collect_stats(const struct cgit_period *period)
{
struct string_list authors;
struct rev_info rev;
struct commit *commit;
// setup_revisions reads the entries after the double dash up to a
// NULL, past the count, so the sentinel has to stay even when the
// path fills the slot before it.
const char *argv[] = {NULL, ctx.qry.head, NULL, NULL, NULL};
int argc = 2;
time_t since;
struct tm tm;
window_start(period, &tm);
since = timegm(&tm);
if (ctx.qry.path) {
argv[2] = "--";
argv[3] = ctx.qry.path;
argc += 2;
}
repo_init_revisions(the_repository, &rev, NULL);
rev.abbrev = DEFAULT_ABBREV;
rev.commit_format = CMIT_FMT_DEFAULT;
rev.max_parents = 1;
rev.verbose_header = 1;
rev.show_root_diff = 0;
// setup_revisions reads argv the way main does and ignores the first
// entry, so the head to walk sits at argv[1].
setup_revisions(argc, argv, &rev, NULL);
// Prune the walk to the displayed window instead of traversing the
// whole history and discarding older commits. The check below still
// bounds the period edge exactly.
rev.max_age = since;
prepare_revision_walk(&rev);
memset(&authors, 0, sizeof(authors));
while ((commit = get_revision(&rev)) != NULL) {
struct commitinfo *info = cgit_parse_commit(commit);
if ((time_t)info->committer_date >= since)
add_commit(&authors, info, period);
cgit_free_commitinfo(info);
release_commit_memory(the_repository->parsed_objects, commit);
commit->parents = NULL;
}
return authors;
}
static int cmp_total_commits(const void *a, const void *b)
{
const struct string_list_item *first = a;
const struct string_list_item *second = b;
const struct authorstat *first_stats = first->util;
const struct authorstat *second_stats = second->util;
// Report the sign only, since a long difference truncated to int
// could flip and leave the comparator inconsistent.
if (second_stats->total > first_stats->total)
return 1;
if (second_stats->total < first_stats->total)
return -1;
return 0;
}
/*
* The column labels for the displayed window, oldest first. pretty hands back
* a buffer it goes on to reuse, so the labels are copied here once and shared
* by every table below.
*/
static struct string_list build_period_labels(const struct cgit_period *period)
{
struct string_list labels = STRING_LIST_INIT_DUP;
struct tm tm;
int i;
window_start(period, &tm);
for (i = 0; i < period->count; i++) {
string_list_append(&labels, period->pretty(&tm));
period->inc(&tm);
}
return labels;
}
/*
* The run of authors to sum is given as a start and a count rather than as
* two indices, so that an empty author list cannot describe a run that wraps.
*/
static void print_summary_row(struct string_list *authors, size_t from,
size_t count, const char *label_format,
const char *leftclass, const char *centerclass,
const char *rightclass,
const struct string_list *labels)
{
struct authorstat *stats;
struct string_list_item *bucket;
size_t i, column;
long total, subtotal;
total = 0;
htmlf("
| %s | ", leftclass,
cgit_fmt(label_format, (long)count));
for (column = 0; column < labels->nr; column++) {
const char *label = labels->items[column].string;
subtotal = 0;
for (i = from; i < from + count; i++) {
stats = authors->items[i].util;
bucket = string_list_lookup(&stats->periods, label);
if (bucket)
subtotal += (uintptr_t)bucket->util;
}
htmlf("%ld | ", centerclass, subtotal);
total += subtotal;
}
htmlf("%ld |
\n", rightclass, total);
}
static void print_authors(struct string_list *authors, int max_rows,
const struct string_list *labels)
{
struct string_list_item *author, *bucket;
struct authorstat *stats;
size_t i, column, rows;
long total;
html("\n| Author | ");
for (column = 0; column < labels->nr; column++)
htmlf("%s | ", labels->items[column].string);
html("Total |
\n");
// The row count arrives through ofs, which carries -1 for "all".
rows = (max_rows <= 0 || (size_t)max_rows > authors->nr)
? authors->nr : (size_t)max_rows;
for (i = 0; i < rows; i++) {
author = &authors->items[i];
html("| ");
html_txt(author->string);
html(" | ");
stats = author->util;
total = 0;
for (column = 0; column < labels->nr; column++) {
const char *label = labels->items[column].string;
bucket = string_list_lookup(&stats->periods, label);
if (!bucket)
html("0 | ");
else {
htmlf("%lu | ", (uintptr_t)bucket->util);
total += (uintptr_t)bucket->util;
}
}
htmlf("%ld |
\n", total);
}
if (rows < authors->nr)
print_summary_row(authors, rows, authors->nr - rows,
"Others (%ld)", "left", "", "sum", labels);
print_summary_row(authors, 0, authors->nr, "Total",
"total", "sum", "sum", labels);
html("
\n");
}
static const char *language_of(const char *pathname)
{
const char *ext;
size_t i;
ext = strrchr(pathname, '.');
if (ext && ext != pathname && ext[1]) {
for (i = 0; i < ARRAY_SIZE(lang_map); i++)
if (!strcasecmp(ext + 1, lang_map[i].ext))
return lang_map[i].label;
} else if (!strcmp(pathname, "Makefile")) {
return "Make";
}
return "Other";
}
static int add_blob_size(const struct object_id *oid, struct strbuf *base,
const char *pathname, unsigned mode, void *data)
{
struct lang_sizes *sizes = data;
struct string_list_item *lang;
unsigned long size;
if (S_ISDIR(mode))
return READ_TREE_RECURSIVE;
if (!S_ISREG(mode))
return 0;
if (odb_read_object_info(the_repository->objects, oid, &size) != OBJ_BLOB
|| !size)
return 0;
lang = string_list_insert(&sizes->langs, language_of(pathname));
lang->util = (void *)((uintptr_t)lang->util + size);
sizes->total += size;
return 0;
}
static int cmp_lang_bytes(const void *a, const void *b)
{
const struct string_list_item *first = a;
const struct string_list_item *second = b;
uintptr_t first_bytes = (uintptr_t)first->util;
uintptr_t second_bytes = (uintptr_t)second->util;
return first_bytes < second_bytes ? 1 :
first_bytes > second_bytes ? -1 : 0;
}
// Leaves the list sorted largest first, which is the order the rows print in
// and what lets the caller stop naming languages once it reaches the tail.
static void measure_languages(struct lang_sizes *sizes)
{
struct pathspec paths = { .nr = 0 };
struct object_id oid;
struct commit *commit;
if (repo_get_oid(the_repository, ctx.qry.head, &oid))
return;
commit = lookup_commit_reference(the_repository, &oid);
if (!commit || repo_parse_commit(the_repository, commit))
return;
read_tree(the_repository, repo_get_commit_tree(the_repository, commit),
&paths, add_blob_size, sizes);
qsort(sizes->langs.items, sizes->langs.nr,
sizeof(struct string_list_item), cmp_lang_bytes);
}
static void print_language_row(const char *label, unsigned long bytes,
unsigned long total)
{
struct strbuf size = STRBUF_INIT;
int tenths;
strbuf_humanise_bytes(&size, bytes);
html("| ");
html_txt(label);
html(" | ");
html_txt(size.buf);
// Printed from integers so a Lua filter switching LC_NUMERIC to a
// comma-decimal locale cannot change the output.
tenths = (int)(1000.0 * bytes / total + 0.5);
htmlf(" | %d.%d%% |
\n", tenths / 10, tenths % 10);
strbuf_release(&size);
}
static void print_languages(const struct lang_sizes *sizes)
{
unsigned long other;
size_t i;
int shown;
if (!sizes->total)
return;
// The Other row is printed last whatever its size, so everything past
// the named languages is summed into it up front.
other = 0;
shown = 0;
for (i = 0; i < sizes->langs.nr; i++) {
if (shown < MAX_LANGUAGE_ROWS &&
strcmp(sizes->langs.items[i].string, "Other")) {
shown++;
continue;
}
other += (uintptr_t)sizes->langs.items[i].util;
}
html("Languages
\n");
html("\n");
html("| Language | Size | Share |
\n");
shown = 0;
for (i = 0; i < sizes->langs.nr && shown < MAX_LANGUAGE_ROWS; i++) {
if (!strcmp(sizes->langs.items[i].string, "Other"))
continue;
print_language_row(sizes->langs.items[i].string,
(uintptr_t)sizes->langs.items[i].util,
sizes->total);
shown++;
}
if (other)
print_language_row("Other", other, sizes->total);
html("
\n");
}
static void print_options_form(const struct cgit_period *period, int top)
{
int choices, i;
html("\n");
html("
stat options");
html("
\n");
html("
\n");
}
int cgit_find_stats_period(const char *expr, const struct cgit_period **period)
{
size_t i;
char code = '\0';
if (!expr)
return 0;
if (strlen(expr) == 1)
code = expr[0];
for (i = 0; i < ARRAY_SIZE(periods); i++)
if (periods[i].code == code || !strcmp(periods[i].name, expr)) {
if (period)
*period = &periods[i];
return i + 1;
}
return 0;
}
const char *cgit_find_stats_periodname(int idx)
{
if (idx > 0 && idx <= (int)ARRAY_SIZE(periods))
return periods[idx - 1].name;
else
return "";
}
void cgit_show_stats(void)
{
struct string_list authors, labels;
struct lang_sizes sizes;
const struct cgit_period *period;
int top, period_index;
const char *code = "w";
if (ctx.qry.period)
code = ctx.qry.period;
period_index = cgit_find_stats_period(code, &period);
if (!period_index) {
cgit_print_error_page(404, "Not found",
"Unknown statistics type: %c", code[0]);
return;
}
if (ctx.repo->max_stats && period_index > ctx.repo->max_stats) {
cgit_print_error_page(400, "Bad request",
"Statistics type disabled: %s", period->name);
return;
}
// The tree has to be measured before the history walk. That walk
// releases each commit as it goes, which resets the commit's slab
// index, so a lookup afterwards would read another commit's slot and
// walk the wrong tree.
memset(&sizes, 0, sizeof(sizes));
measure_languages(&sizes);
authors = collect_stats(period);
qsort(authors.items, authors.nr, sizeof(struct string_list_item),
cmp_total_commits);
top = ctx.qry.ofs;
if (!top)
top = DEFAULT_AUTHOR_ROWS;
cgit_print_layout_start();
print_options_form(period, top);
htmlf("Commits per author per %s", period->name);
if (ctx.qry.path) {
html(" (path '");
html_txt(ctx.qry.path);
html("')");
}
html("
\n");
labels = build_period_labels(period);
print_authors(&authors, top, &labels);
string_list_clear(&labels, 0);
print_languages(&sizes);
string_list_clear(&sizes.langs, 0);
cgit_print_layout_end();
}