From ffd23bfd13a2f009f4c1c6b355a5f895969f81fd Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Mon, 20 Jul 2026 10:18:48 -1000 Subject: Gate the stats page and add a language breakdown `max-stats` only bounds the selectable periods now and no longer doubles as the enable switch. The tree walk runs before the history walk on purpose. Releasing commit memory while walking history resets each commit slab index, and a commit graph lookup afterwards would read another commit slot and walk the wrong tree. The stats fixture writes a commit graph so the tests cover that path. The history walk bounds the window in process rather than passing a formatted since date to `setup_revisions`, and parses each commit once. --- source/ui-stats.c | 200 +++++++++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 182 insertions(+), 18 deletions(-) (limited to 'source/ui-stats.c') diff --git a/source/ui-stats.c b/source/ui-stats.c index cabbc3c..f6d2637 100644 --- a/source/ui-stats.c +++ b/source/ui-stats.c @@ -168,10 +168,9 @@ const char *cgit_find_stats_periodname(int idx) return ""; } -static void add_commit(struct string_list *authors, struct commit *commit, +static void add_commit(struct string_list *authors, struct commitinfo *info, const struct cgit_period *period) { - struct commitinfo *info; struct string_list_item *author, *item; struct authorstat *authorstat; struct string_list *items; @@ -180,7 +179,6 @@ static void add_commit(struct string_list *authors, struct commit *commit, time_t t; uintptr_t *counter; - info = cgit_parse_commit(commit); /* A commit can lack an author header, so fall back rather than * xstrdup(NULL). */ tmp = xstrdup(info->author ? info->author : "(unknown)"); @@ -202,7 +200,6 @@ static void add_commit(struct string_list *authors, struct commit *commit, (*counter)++; authorstat->total++; - cgit_free_commitinfo(info); } static int cmp_total_commits(const void *a1, const void *a2) @@ -215,31 +212,31 @@ static int cmp_total_commits(const void *a1, const void *a2) return auth2->total - auth1->total; } -/* Walk the commit DAG and collect number of commits per author per - * timeperiod into a nested string_list collection. +/* Walk the commit DAG once for the configured window. One walk, no + * diffs, so the cost class matches rev-list and the page cache absorbs + * repeat views. Merge commits are skipped. */ static struct string_list collect_stats(const struct cgit_period *period) { struct string_list authors; struct rev_info rev; struct commit *commit; - const char *argv[] = {NULL, ctx.qry.head, NULL, NULL, NULL, NULL}; - int argc = 3; - time_t now; + const char *argv[] = {NULL, ctx.qry.head, NULL, NULL}; + int argc = 2; + time_t now, since; long i; struct tm tm; - char tmp[11]; time(&now); gmtime_r(&now, &tm); period->trunc(&tm); for (i = 1; i < period->count; i++) period->dec(&tm); - strftime(tmp, sizeof(tmp), "%Y-%m-%d", &tm); - argv[2] = xstrdup(fmt("--since=%s", tmp)); + since = timegm(&tm); + if (ctx.qry.path) { - argv[3] = "--"; - argv[4] = ctx.qry.path; + argv[2] = "--"; + argv[3] = ctx.qry.path; argc += 2; } repo_init_revisions(the_repository, &rev, NULL); @@ -252,7 +249,12 @@ static struct string_list collect_stats(const struct cgit_period *period) prepare_revision_walk(&rev); memset(&authors, 0, sizeof(authors)); while ((commit = get_revision(&rev)) != NULL) { - add_commit(&authors, commit, period); + struct commitinfo *info = cgit_parse_commit(commit); + + if (info->committer_date >= since) + add_commit(&authors, info, period); + + cgit_free_commitinfo(info); release_commit_memory(the_repository->parsed_objects, commit); commit->parents = NULL; } @@ -364,6 +366,150 @@ static void print_authors(struct string_list *authors, int top, html(""); } +/* Bytes at HEAD per language, judged by file extension. One recursive + * tree read, sizes come from object headers without loading content. */ +static const struct { + const char *ext; + const char *label; +} lang_map[] = { + {"c", "C"}, {"h", "C"}, + {"cpp", "C++"}, {"cc", "C++"}, {"cxx", "C++"}, + {"hpp", "C++"}, {"hh", "C++"}, + {"js", "JavaScript"}, {"mjs", "JavaScript"}, + {"ts", "TypeScript"}, {"tsx", "TypeScript"}, + {"py", "Python"}, {"lua", "Lua"}, + {"sh", "Shell"}, {"bash", "Shell"}, + {"go", "Go"}, {"rs", "Rust"}, {"zig", "Zig"}, + {"java", "Java"}, {"kt", "Kotlin"}, {"cs", "C#"}, {"swift", "Swift"}, + {"rb", "Ruby"}, {"pl", "Perl"}, {"pm", "Perl"}, {"php", "PHP"}, + {"hs", "Haskell"}, {"el", "Lisp"}, {"ml", "OCaml"}, + {"css", "CSS"}, {"scss", "CSS"}, + {"html", "HTML"}, {"htm", "HTML"}, {"xml", "XML"}, + {"md", "Markdown"}, {"rst", "Text"}, {"txt", "Text"}, + {"json", "JSON"}, {"yml", "YAML"}, {"yaml", "YAML"}, {"toml", "TOML"}, + {"mk", "Make"}, {"tex", "TeX"}, {"sql", "SQL"}, {"vim", "Vimscript"}, +}; + +struct lang_walk_ctx { + struct string_list langs; + unsigned long total; +}; + +static int lang_walk_cb(const struct object_id *oid, struct strbuf *base, + const char *pathname, unsigned mode, void *cbdata) +{ + struct lang_walk_ctx *lw = cbdata; + struct string_list_item *item; + const char *ext, *label = NULL; + unsigned long size; + int i; + + if (S_ISDIR(mode)) + return READ_TREE_RECURSIVE; + if (!S_ISREG(mode)) + return 0; + if (odb_read_object_info(the_repository->objects, oid, &size) != OBJ_BLOB + || !size) + return 0; + + ext = strrchr(pathname, '.'); + if (ext && ext != pathname && ext[1]) { + for (i = 0; i < (int)ARRAY_SIZE(lang_map); i++) + if (!strcasecmp(ext + 1, lang_map[i].ext)) { + label = lang_map[i].label; + break; + } + } else if (!strcmp(pathname, "Makefile")) { + label = "Make"; + } + if (!label) + label = "Other"; + item = string_list_insert(&lw->langs, label); + item->util = (void *)((uintptr_t)item->util + size); + lw->total += size; + return 0; +} + +static int cmp_lang_bytes(const void *a1, const void *a2) +{ + const struct string_list_item *i1 = a1; + const struct string_list_item *i2 = a2; + uintptr_t b1 = (uintptr_t)i1->util; + uintptr_t b2 = (uintptr_t)i2->util; + + return b1 < b2 ? 1 : b1 > b2 ? -1 : 0; +} + +static void summarize_tree(struct lang_walk_ctx *lw) +{ + struct pathspec paths = { .nr = 0 }; + struct object_id oid; + struct commit *commit; + + if (repo_get_oid(the_repository, ctx.qry.head, &oid)) + return; + commit = lookup_commit_reference(the_repository, &oid); + if (!commit || repo_parse_commit(the_repository, commit)) + return; + + read_tree(the_repository, repo_get_commit_tree(the_repository, commit), + &paths, lang_walk_cb, lw); + qsort(lw->langs.items, lw->langs.nr, sizeof(struct string_list_item), + cmp_lang_bytes); +} + +static void print_language_row(const char *label, unsigned long bytes, + unsigned long total) +{ + struct strbuf size = STRBUF_INIT; + + strbuf_humanise_bytes(&size, bytes); + html(""); + html_txt(label); + html(""); + html_txt(size.buf); + htmlf("%.1f%%\n", 100.0 * bytes / total); + strbuf_release(&size); +} + +/* Bordered like the commits-per-author table. */ +static void print_languages(struct lang_walk_ctx *lw) +{ + unsigned long other; + int i, shown; + + if (!lw->total) + return; + + /* Show up to six named rows, everything else folds into an + * Other row at the end. */ + other = 0; + shown = 0; + for (i = 0; i < lw->langs.nr; i++) { + if (shown < 6 && strcmp(lw->langs.items[i].string, "Other")) { + shown++; + continue; + } + other += (uintptr_t)lw->langs.items[i].util; + } + + html("

Languages

"); + html(""); + html("\n"); + shown = 0; + for (i = 0; i < lw->langs.nr && shown < 6; i++) { + if (!strcmp(lw->langs.items[i].string, "Other")) + continue; + print_language_row(lw->langs.items[i].string, + (uintptr_t)lw->langs.items[i].util, + lw->total); + shown++; + } + if (other) + print_language_row("Other", other, lw->total); + html("
LanguageSizeShare
"); +} + /* Create a sorted string_list with one entry per author. The util-field * for each author is another string_list which is used to calculate the * number of commits per time-interval. @@ -371,6 +517,7 @@ static void print_authors(struct string_list *authors, int top, void cgit_show_stats(void) { struct string_list authors; + struct lang_walk_ctx lw; const struct cgit_period *period; int top, i; const char *code = "w"; @@ -384,11 +531,18 @@ void cgit_show_stats(void) "Unknown statistics type: %c", code[0]); return; } - if (i > ctx.repo->max_stats) { + if (ctx.repo->max_stats && i > ctx.repo->max_stats) { cgit_print_error_page(400, "Bad request", "Statistics type disabled: %s", period->name); return; } + /* Walk the tree before the history walk. Releasing commit memory + * during that walk resets each commit's slab index, and a later + * lookup through the commit graph would then read another + * commit's slot and walk the wrong tree. */ + memset(&lw, 0, sizeof(lw)); + summarize_tree(&lw); + authors = collect_stats(period); qsort(authors.items, authors.nr, sizeof(struct string_list_item), cmp_total_commits); @@ -398,15 +552,20 @@ void cgit_show_stats(void) top = 10; cgit_print_layout_start(); + + /* The options panel floats right of the page top, the same spot + * the diff controls occupy on the diff pages. */ html("
"); html("stat options"); html("
"); cgit_add_hidden_formfields(1, 0, "stats"); html(""); - if (ctx.repo->max_stats > 1) { + if (!ctx.repo->max_stats || ctx.repo->max_stats > 1) { + int nperiods = ctx.repo->max_stats ? + ctx.repo->max_stats : (int)ARRAY_SIZE(periods); html(""); html(""); @@ -424,6 +583,7 @@ void cgit_show_stats(void) html("
Period:
"); html("
"); html("
"); + htmlf("

Commits per author per %s", period->name); if (ctx.qry.path) { html(" (path '"); @@ -432,6 +592,10 @@ void cgit_show_stats(void) } html("

"); print_authors(&authors, top, period); + + print_languages(&lw); + string_list_clear(&lw.langs, 0); + cgit_print_layout_end(); } -- cgit v2.8.0