commit-reach: make can_all_from_reach... linear
authorDerrick Stolee <dstolee@microsoft.com>
Fri, 20 Jul 2018 16:33:28 +0000 (16:33 +0000)
committerJunio C Hamano <gitster@pobox.com>
Fri, 20 Jul 2018 22:38:56 +0000 (15:38 -0700)
The can_all_from_reach_with_flags() algorithm is currently quadratic in
the worst case, because it calls the reachable() method for every 'from'
without tracking which commits have already been walked or which can
already reach a commit in 'to'.

Rewrite the algorithm to walk each commit a constant number of times.

We also add some optimizations that should work for the main consumer of
this method: fetch negotitation (haves/wants).

The first step includes using a depth-first-search (DFS) from each
'from' commit, sorted by ascending generation number. We do not walk
beyond the minimum generation number or the minimum commit date. This
DFS is likely to be faster than the existing reachable() method because
we expect previous ref values to be along the first-parent history.

If we find a target commit, then we mark everything in the DFS stack as
a RESULT. This expands the set of targets for the other 'from' commits.
We also mark the visited commits using 'assign_flag' to prevent re-
walking the same commits.

We still need to clear our flags at the end, which is why we will have a
total of three visits to each commit.

Performance was measured on the Linux repository using
'test-tool reach can_all_from_reach'. The input included rows seeded by
tag values. The "small" case included X-rows as v4.[0-9]* and Y-rows as
v3.[0-9]*. This mimics a (very large) fetch that says "I have all major
v3 releases and want all major v4 releases." The "large" case included
X-rows as "v4.*" and Y-rows as "v3.*". This adds all release-candidate
tags to the set, which does not greatly increase the number of objects
that are considered, but does increase the number of 'from' commits,
demonstrating the quadratic nature of the previous code.

Small Case:

Before: 1.52 s
After: 0.26 s

Large Case:

Before: 3.50 s
After: 0.27 s

Note how the time increases between the two cases in the two versions.
The new code increases relative to the number of commits that need to be
walked, but not directly relative to the number of 'from' commits.

Signed-off-by: Derrick Stolee <dstolee@microsoft.com>
Signed-off-by: Junio C Hamano <gitster@pobox.com>
commit-reach.c
commit-reach.h
upload-pack.c
index f5858944fda9fac5159367fafe6f8085a903c168..bc522d6840f3c9fc4ae79adb0f2fc339fe37df38 100644 (file)
@@ -514,66 +514,87 @@ int commit_contains(struct ref_filter *filter, struct commit *commit,
        return is_descendant_of(commit, list);
 }
 
-int reachable(struct commit *from, unsigned int with_flag,
-             unsigned int assign_flag, time_t min_commit_date)
+static int compare_commits_by_gen(const void *_a, const void *_b)
 {
-       struct prio_queue work = { compare_commits_by_commit_date };
+       const struct commit *a = (const struct commit *)_a;
+       const struct commit *b = (const struct commit *)_b;
 
-       prio_queue_put(&work, from);
-       while (work.nr) {
-               struct commit_list *list;
-               struct commit *commit = prio_queue_get(&work);
-
-               if (commit->object.flags & with_flag) {
-                       from->object.flags |= assign_flag;
-                       break;
-               }
-               if (!commit->object.parsed)
-                       parse_object(the_repository, &commit->object.oid);
-               if (commit->object.flags & REACHABLE)
-                       continue;
-               commit->object.flags |= REACHABLE;
-               if (commit->date < min_commit_date)
-                       continue;
-               for (list = commit->parents; list; list = list->next) {
-                       struct commit *parent = list->item;
-                       if (!(parent->object.flags & REACHABLE))
-                               prio_queue_put(&work, parent);
-               }
-       }
-       from->object.flags |= REACHABLE;
-       clear_commit_marks(from, REACHABLE);
-       clear_prio_queue(&work);
-       return (from->object.flags & assign_flag);
+       if (a->generation < b->generation)
+               return -1;
+       if (a->generation > b->generation)
+               return 1;
+       return 0;
 }
 
 int can_all_from_reach_with_flag(struct object_array *from,
                                 unsigned int with_flag,
                                 unsigned int assign_flag,
-                                time_t min_commit_date)
+                                time_t min_commit_date,
+                                uint32_t min_generation)
 {
+       struct commit **list = NULL;
        int i;
+       int result = 1;
 
+       ALLOC_ARRAY(list, from->nr);
        for (i = 0; i < from->nr; i++) {
-               struct object *from_one = from->objects[i].item;
+               list[i] = (struct commit *)from->objects[i].item;
 
-               if (from_one->flags & assign_flag)
-                       continue;
-               from_one = deref_tag(the_repository, from_one, "a from object", 0);
-               if (!from_one || from_one->type != OBJ_COMMIT) {
-                       /* no way to tell if this is reachable by
-                        * looking at the ancestry chain alone, so
-                        * leave a note to ourselves not to worry about
-                        * this object anymore.
-                        */
-                       from->objects[i].item->flags |= assign_flag;
-                       continue;
-               }
-               if (!reachable((struct commit *)from_one, with_flag, assign_flag,
-                              min_commit_date))
+               if (parse_commit(list[i]) ||
+                   list[i]->generation < min_generation)
                        return 0;
        }
-       return 1;
+
+       QSORT(list, from->nr, compare_commits_by_gen);
+
+       for (i = 0; i < from->nr; i++) {
+               /* DFS from list[i] */
+               struct commit_list *stack = NULL;
+
+               list[i]->object.flags |= assign_flag;
+               commit_list_insert(list[i], &stack);
+
+               while (stack) {
+                       struct commit_list *parent;
+
+                       if (stack->item->object.flags & with_flag) {
+                               pop_commit(&stack);
+                               continue;
+                       }
+
+                       for (parent = stack->item->parents; parent; parent = parent->next) {
+                               if (parent->item->object.flags & (with_flag | RESULT))
+                                       stack->item->object.flags |= RESULT;
+
+                               if (!(parent->item->object.flags & assign_flag)) {
+                                       parent->item->object.flags |= assign_flag;
+
+                                       if (parse_commit(parent->item) ||
+                                           parent->item->date < min_commit_date ||
+                                           parent->item->generation < min_generation)
+                                               continue;
+
+                                       commit_list_insert(parent->item, &stack);
+                                       break;
+                               }
+                       }
+
+                       if (!parent)
+                               pop_commit(&stack);
+               }
+
+               if (!(list[i]->object.flags & (with_flag | RESULT))) {
+                       result = 0;
+                       goto cleanup;
+               }
+       }
+
+cleanup:
+       for (i = 0; i < from->nr; i++) {
+               clear_commit_marks(list[i], RESULT);
+               clear_commit_marks(list[i], assign_flag);
+       }
+       return result;
 }
 
 int can_all_from_reach(struct commit_list *from, struct commit_list *to,
@@ -583,6 +604,7 @@ int can_all_from_reach(struct commit_list *from, struct commit_list *to,
        time_t min_commit_date = cutoff_by_min_date ? from->item->date : 0;
        struct commit_list *from_iter = from, *to_iter = to;
        int result;
+       uint32_t min_generation = GENERATION_NUMBER_INFINITY;
 
        while (from_iter) {
                add_object_array(&from_iter->item->object, NULL, &from_objs);
@@ -590,6 +612,9 @@ int can_all_from_reach(struct commit_list *from, struct commit_list *to,
                if (!parse_commit(from_iter->item)) {
                        if (from_iter->item->date < min_commit_date)
                                min_commit_date = from_iter->item->date;
+
+                       if (from_iter->item->generation < min_generation)
+                               min_generation = from_iter->item->generation;
                }
 
                from_iter = from_iter->next;
@@ -599,6 +624,9 @@ int can_all_from_reach(struct commit_list *from, struct commit_list *to,
                if (!parse_commit(to_iter->item)) {
                        if (to_iter->item->date < min_commit_date)
                                min_commit_date = to_iter->item->date;
+
+                       if (to_iter->item->generation < min_generation)
+                               min_generation = to_iter->item->generation;
                }
 
                to_iter->item->object.flags |= PARENT2;
@@ -607,7 +635,7 @@ int can_all_from_reach(struct commit_list *from, struct commit_list *to,
        }
 
        result = can_all_from_reach_with_flag(&from_objs, PARENT2, PARENT1,
-                                             min_commit_date);
+                                             min_commit_date, min_generation);
 
        while (from) {
                clear_commit_marks(from->item, PARENT1);
index aa202c97038e76ea899277338563b2b53ff75615..7d313e2975a28ef79b2ccf6dea965d4f139eda58 100644 (file)
@@ -59,19 +59,18 @@ define_commit_slab(contains_cache, enum contains_result);
 int commit_contains(struct ref_filter *filter, struct commit *commit,
                    struct commit_list *list, struct contains_cache *cache);
 
-int reachable(struct commit *from, unsigned int with_flag,
-             unsigned int assign_flag, time_t min_commit_date);
-
 /*
  * Determine if every commit in 'from' can reach at least one commit
  * that is marked with 'with_flag'. As we traverse, use 'assign_flag'
  * as a marker for commits that are already visited. Do not walk
- * commits with date below 'min_commit_date'.
+ * commits with date below 'min_commit_date' or generation below
+ * 'min_generation'.
  */
 int can_all_from_reach_with_flag(struct object_array *from,
                                 unsigned int with_flag,
                                 unsigned int assign_flag,
-                                time_t min_commit_date);
+                                time_t min_commit_date,
+                                uint32_t min_generation);
 int can_all_from_reach(struct commit_list *from, struct commit_list *to,
                       int commit_date_cutoff);
 
index 11c426685d2f2ba7c50d798b6ace6e619a265aed..1e498f1188c90e4998bb06e17f99c4780592a553 100644 (file)
@@ -338,11 +338,14 @@ static int got_oid(const char *hex, struct object_id *oid)
 
 static int ok_to_give_up(void)
 {
+       uint32_t min_generation = GENERATION_NUMBER_ZERO;
+
        if (!have_obj.nr)
                return 0;
 
        return can_all_from_reach_with_flag(&want_obj, THEY_HAVE,
-                                           COMMON_KNOWN, oldest_have);
+                                           COMMON_KNOWN, oldest_have,
+                                           min_generation);
 }
 
 static int get_common_commits(void)