package support; import java.util.*; /** * MongodbIndexPlannerAlgorithm — models CWE-407 in MongoDB query planner RelevantTag. * * ── mongodb-0001 planner_ixselect.cpp (CRITICAL — 4 sites) ────────────────── * * DEFECT: RelevantTag in index_tag.h stores index assignments in two * std::vector members: * * std::vector first; // index_tag.h:106 * std::vector notFirst; // index_tag.h:107 * * Three operations use std::find on these vectors — O(n) each: * * Site 1: removeIndexRelevantTag() at ~planner_ixselect.cpp:972 * vector::iterator firstIt = std::find(tag->first.begin(), tag->first.end(), idx); * if (firstIt != tag->first.end()) tag->first.erase(firstIt); * // repeated for notFirst * * Sites 2/3/4: stripInvalidAssignments*() at lines 1084/1086, 1310/1313, 1424/1427 * bool inFirst = tag->first.end() != std::find(tag->first.begin(), tag->first.end(), idx); * bool inNotFirst = tag->notFirst.end() != std::find(tag->notFirst.begin(), ...); * * These are called in loops over every node in the query match expression tree. * With N match nodes and M relevant indexes per tag, each strip pass is O(N × M). * * Fix: Change RelevantTag::first and RelevantTag::notFirst from std::vector * to std::unordered_set. Single struct change. All four sites become O(1): * - contains check: .count(idx) > 0 — O(1) * - remove: .erase(idx) — O(1) * - insert: .insert(i) instead of push_back(i) — O(1) * * Comparison counts (N nodes, M indexes per tag): * defective: N × M (M-element std::find scan per node per strip pass) * fixed: N (O(1) hash lookup per node per strip pass) * * ── mongodb-0003 plan_enumerator.cpp (HIGH — 4 sites) ─────────────────────── * * DEFECT: plan_enumerator.cpp maintains tracked predicate vectors for de-dup: * * Site 697: std::find on assignedPreds vector * Site 734: std::find on predsAssigned * Site 753: std::find on predsIntersected * Site 816: std::find on subnodePreds * * Same pattern: O(n) membership check inside a loop over predicates. * With P predicates and A already-assigned entries: O(P × A). * * Fix: Replace each vector with unordered_set. * Pointer identity is correct — each MatchExpression* is a unique predicate node. * * Comparison counts (P predicates, A already-assigned): * defective: P × A (each new predicate scans all assigned) * fixed: P (one O(1) hash lookup per predicate) */ public class MongodbIndexPlannerAlgorithm { // ── RelevantTag model ───────────────────────────────────────────────────── // // In real MongoDB, first/notFirst store index numbers (size_t). // We model them as integers. The critical property is membership check + remove. /** Defective RelevantTag: vector — O(n) scan for contains and remove. */ public static final class DefectiveRelevantTag { public final List first = new ArrayList<>(); public final List notFirst = new ArrayList<>(); /** models: push_back(i) at planner_ixselect.cpp:819/821 */ public void assign(int idx, boolean isFirst) { if (isFirst) first.add(idx); else notFirst.add(idx); } /** * models: removeIndexRelevantTag — std::find + erase, O(n). * Returns number of comparisons made. */ public long remove(int idx) { long cmp = 0; // std::find on first for (int i = 0; i < first.size(); i++) { cmp++; if (first.get(i) == idx) { first.remove(i); break; } } // std::find on notFirst for (int i = 0; i < notFirst.size(); i++) { cmp++; if (notFirst.get(i) == idx) { notFirst.remove(i); break; } } return cmp; } /** * models: inFirst || inNotFirst membership check at lines 1084,1310,1424. * Returns [result, comparisons]. */ public long[] contains(int idx) { long cmp = 0; boolean found = false; for (int v : first) { cmp++; if (v == idx) { found = true; break; } } if (!found) { for (int v : notFirst) { cmp++; if (v == idx) { found = true; break; } } } return new long[]{ found ? 1 : 0, cmp }; } } /** Fixed RelevantTag: unordered_set — O(1) for all operations. */ public static final class FixedRelevantTag { public final Set first = new HashSet<>(); public final Set notFirst = new HashSet<>(); /** models: .insert(i) — O(1) */ public void assign(int idx, boolean isFirst) { if (isFirst) first.add(idx); else notFirst.add(idx); } /** models: .erase(idx) — O(1). Returns comparisons = 1. */ public long remove(int idx) { first.remove(idx); notFirst.remove(idx); return 1; // one O(1) hash op each } /** models: .count(idx) > 0 — O(1). Returns [result, comparisons=1]. */ public long[] contains(int idx) { boolean found = first.contains(idx) || notFirst.contains(idx); return new long[]{ found ? 1 : 0, 1 }; } } // ── Strip-invalid-assignments simulation ───────────────────────────────── // // Models the strip passes: for each node in the match tree, check if it is // assigned to a particular index. If yes, strip it. N nodes × M indexes. public static final class StripResult { public final int strippedCount; public final long comparisons; public StripResult(int strippedCount, long comparisons) { this.strippedCount = strippedCount; this.comparisons = comparisons; } } /** * Defective: O(N × M) — std::find scan on vector per node per index check. * * @param tags simulated match-expression nodes, each with a DefectiveRelevantTag * @param idxToStrip the index number whose assignments should be removed */ public static StripResult defectiveStripInvalidAssignments( List tags, int idxToStrip) { long comparisons = 0; int stripped = 0; for (DefectiveRelevantTag tag : tags) { long[] result = tag.contains(idxToStrip); comparisons += result[1]; if (result[0] != 0) { comparisons += tag.remove(idxToStrip); stripped++; } } return new StripResult(stripped, comparisons); } /** * Fixed: O(N) — O(1) hash lookup per node. * * @param tags simulated match-expression nodes, each with a FixedRelevantTag * @param idxToStrip the index number whose assignments should be removed */ public static StripResult fixedStripInvalidAssignments( List tags, int idxToStrip) { long comparisons = 0; int stripped = 0; for (FixedRelevantTag tag : tags) { long[] result = tag.contains(idxToStrip); comparisons += result[1]; if (result[0] != 0) { comparisons += tag.remove(idxToStrip); stripped++; } } return new StripResult(stripped, comparisons); } // ── PlanEnumerator predicate de-dup model (mongodb-0003) ───────────────── public static final class EnumResult { public final Set assigned; public final long comparisons; public EnumResult(Set assigned, long comparisons) { this.assigned = assigned; this.comparisons = comparisons; } } /** * Defective: std::find on assignedPreds vector — O(A) per predicate. * A = size of already-assigned set at the time of each check. */ public static EnumResult defectiveEnumeratePreds(List predicates) { List assigned = new ArrayList<>(); long comparisons = 0; for (int pred : predicates) { // std::find scan boolean found = false; for (int a : assigned) { comparisons++; if (a == pred) { found = true; break; } } if (!found) assigned.add(pred); } return new EnumResult(new HashSet<>(assigned), comparisons); } /** * Fixed: std::unordered_set — O(1) per predicate. */ public static EnumResult fixedEnumeratePreds(List predicates) { Set assigned = new HashSet<>(); long comparisons = 0; for (int pred : predicates) { comparisons++; // O(1) hash lookup assigned.add(pred); } return new EnumResult(assigned, comparisons); } // ── Test data builders ──────────────────────────────────────────────────── /** * Builds N defective tags, each assigned M indexes uniformly spread across * [0, totalIndexes). The tag at position targetFraction% gets idxToStrip assigned. */ public static List buildDefectiveTags( int numNodes, int indexesPerTag, int totalIndexes, Random rng) { List tags = new ArrayList<>(); for (int n = 0; n < numNodes; n++) { DefectiveRelevantTag tag = new DefectiveRelevantTag(); for (int m = 0; m < indexesPerTag; m++) { int idx = rng.nextInt(totalIndexes); tag.assign(idx, rng.nextBoolean()); } tags.add(tag); } return tags; } /** Builds the equivalent fixed tags with identical index assignments. */ public static List buildFixedTags( int numNodes, int indexesPerTag, int totalIndexes, Random rng) { // Use same seed to produce identical assignments List tags = new ArrayList<>(); for (int n = 0; n < numNodes; n++) { FixedRelevantTag tag = new FixedRelevantTag(); for (int m = 0; m < indexesPerTag; m++) { int idx = rng.nextInt(totalIndexes); tag.assign(idx, rng.nextBoolean()); } tags.add(tag); } return tags; } // ── Self-test ───────────────────────────────────────────────────────────── public static void main(String[] args) { System.out.println("MongodbIndexPlannerAlgorithm — mongodb-0001 (planner_ixselect.cpp) + mongodb-0003 (plan_enumerator.cpp)"); System.out.println(); // 0001: strip-invalid-assignments pass System.out.println("── mongodb-0001: RelevantTag strip pass (4 sites in planner_ixselect.cpp) ──"); System.out.println("Defect: std::vector first/notFirst — std::find O(M) per node"); System.out.println("Fix: std::unordered_set first/notFirst — .count() O(1) per node"); System.out.println(); System.out.printf("%-10s %-10s %-14s %-10s %s%n", "N nodes", "M indexes", "Defective ops", "Fixed ops", "Speedup"); System.out.println("─".repeat(58)); int[] nodeCounts = {10, 50, 100, 500, 1000, 5000}; int idxPerTag = 20; int totalIdxs = 30; for (int n : nodeCounts) { Random rng1 = new Random(42); Random rng2 = new Random(42); // same seed List defTags = buildDefectiveTags(n, idxPerTag, totalIdxs, rng1); List fixTags = buildFixedTags(n, idxPerTag, totalIdxs, rng2); StripResult def = defectiveStripInvalidAssignments(defTags, 5); StripResult fix = fixedStripInvalidAssignments(fixTags, 5); assert def.strippedCount == fix.strippedCount : "stripped count mismatch at N=" + n; System.out.printf("%-10d %-10d %-14d %-10d %.1fx%n", n, idxPerTag, def.comparisons, fix.comparisons, (double) def.comparisons / Math.max(1, fix.comparisons)); } System.out.println(); // 0003: plan enumerator predicate de-dup System.out.println("── mongodb-0003: plan_enumerator predicate de-dup (4 sites) ──"); System.out.println("Defect: std::vector — std::find O(A) per predicate"); System.out.println("Fix: std::unordered_set — O(1) per predicate"); System.out.println(); System.out.printf("%-12s %-14s %-10s %s%n", "P predicates", "Defective ops", "Fixed ops", "Speedup"); System.out.println("─".repeat(46)); Random predRng = new Random(42); for (int p : new int[]{10, 50, 100, 500, 1000}) { List preds = new ArrayList<>(); for (int i = 0; i < p; i++) preds.add(i); for (int i = 0; i < p / 4; i++) preds.add(predRng.nextInt(p)); // revisits Collections.shuffle(preds, predRng); EnumResult def = defectiveEnumeratePreds(preds); EnumResult fix = fixedEnumeratePreds(preds); assert def.assigned.equals(fix.assigned) : "assigned set mismatch at P=" + p; System.out.printf("%-12d %-14d %-10d %.1fx%n", preds.size(), def.comparisons, fix.comparisons, (double) def.comparisons / Math.max(1, fix.comparisons)); } } }