Load DBLP data once instead of reloading per index config

The Manager previously called setupBoth() on every test step, which
dropped both PostgreSQL and MariaDB tables and re-imported the full
DBLP TSVs (~9 reloads across 2 DBs in a single run, even though only
PostgreSQL was used). Now data is loaded once via load_data_postgres()
and only the indexes are swapped between tests via apply_indexes_postgres().

A reset_postgres(config, reload_data=False) helper lets callers force
a full reload when needed -- used after 'cl-both' tests, since CLUSTER
physically reorders the table and dropping the index alone does not
restore the original layout.

Also fixes two bugs found while reading the code:
- baseStrategy.run iterated 'self.queries' instead of '.items()',
  which would crash on the first unpack.
- main.py was missing an execute() call after the Aufgabe 3 'cl-both'
  setup, silently skipping that test.

Co-Authored-By: Claude Opus 4.7 <[email protected]>
This commit is contained in:
Sandro Fuetsch Titan
2026-05-26 19:50:07 +02:00
co-authored by Claude Opus 4.7
parent 8a8926cef8
commit d8a906e2d4
4 changed files with 101 additions and 63 deletions
+1 -1
View File
@@ -17,7 +17,7 @@ class BaseStrategy:
self.queries = queries self.queries = queries
def run(self): def run(self):
for strategy, query in self.queries: for strategy, query in self.queries.items():
print("Running: " + strategy + "\n") print("Running: " + strategy + "\n")
start = time.time() start = time.time()
self.cursor.execute(query) self.cursor.execute(query)
+47 -49
View File
@@ -1,49 +1,47 @@
from manager import Manager from manager import Manager
from baseStrategy import BaseStrategy from baseStrategy import BaseStrategy
from nestedInnerLoopStrategy import NestedInnerLoopStrategy from nestedInnerLoopStrategy import NestedInnerLoopStrategy
from sortMergeStrategy import SortMergeStrategy from sortMergeStrategy import SortMergeStrategy
from hashJoinStrategy import HashJoinStrategy from hashJoinStrategy import HashJoinStrategy
from setup import get_connection from setup import get_connection, load_data_postgres
from repositories.all_queries import queries_postgres from repositories.all_queries import queries_postgres
if __name__ == '__main__':
join_manager: Manager | None = None if __name__ == '__main__':
db_post, conn_postsql = get_connection(maria=False)
db_post, conn_postsql = get_connection(maria=False) # Daten einmalig laden; danach werden zwischen den Tests nur die Indexe getauscht.
queries_explict_no_index_postgresql = queries_postgres["no_index"] load_data_postgres()
# Aufgabe 1
#Test on Postgresql join_manager = Manager(BaseStrategy(conn_postsql, db_post, queries_postgres["with_index"]))
#Aufgabe 1 # join_manager.setup_db("no-index")
join_manager = Manager(BaseStrategy(conn_postsql, db_post, queries_postgres["with_index"])) # join_manager.execute() # 3 Mrd. Tupel via Kreuzprodukt -> > 10 min
#join_manager.setup_db("no-index") join_manager.setup_db("unique-publ")
#join_manager.setQueries(queries_ignore_index) 3 Billionen Einträge durch kreuzprodukt > 10min join_manager.execute()
#join_manager.execute() join_manager.setup_db("cl-both")
join_manager.setup_db("unique-publ") join_manager.execute()
join_manager.execute()
join_manager.setup_db("cl-both") # Aufgabe 2 — vorheriger Lauf war 'cl-both' (CLUSTER hat Tabelle physisch sortiert);
join_manager.execute() # reload_data=True stellt die ursprüngliche Ladereihenfolge wieder her.
#Aufgabe 2 join_manager.setStrategy(NestedInnerLoopStrategy(conn_postsql, db_post, queries_postgres["with_index"]))
join_manager.setStrategy(NestedInnerLoopStrategy(conn_postsql, db_post, queries_postgres["with_index"])) join_manager.setup_db("nc-publ", reload_data=True)
join_manager.setup_db("nc-publ") join_manager.execute()
join_manager.execute() join_manager.setup_db("nc-auth")
join_manager.setup_db("nc-auth") join_manager.execute()
join_manager.execute() join_manager.setup_db("nc-both")
join_manager.setup_db("nc-both") join_manager.execute()
join_manager.execute()
#Aufgabe 3 # Aufgabe 3 — Tabelle ist noch im Original-Layout, kein Reload nötig.
join_manager.setStrategy(SortMergeStrategy(conn_postsql, db_post, queries_postgres["no_index"])) join_manager.setStrategy(SortMergeStrategy(conn_postsql, db_post, queries_postgres["no_index"]))
join_manager.setup_db("no-index") join_manager.setup_db("no-index")
join_manager.execute() join_manager.execute()
join_manager.setQueries(queries_postgres["with_index"]) join_manager.setQueries(queries_postgres["with_index"])
join_manager.setup_db("nc-both") join_manager.setup_db("nc-both")
join_manager.execute() join_manager.execute()
join_manager.setup_db("cl-both") join_manager.setup_db("cl-both")
#Aufgabe 4 join_manager.execute()
join_manager.setStrategy(HashJoinStrategy(conn_postsql, db_post, queries_postgres["no_index"]))
join_manager.setup_db("no-index") # Aufgabe 4 — wieder vom CLUSTER-Zustand wegkommen.
join_manager.execute() join_manager.setStrategy(HashJoinStrategy(conn_postsql, db_post, queries_postgres["no_index"]))
join_manager.setup_db("no-index", reload_data=True)
join_manager.execute()
+2 -2
View File
@@ -12,8 +12,8 @@ class Manager:
self.strategy.set_queries(queries) self.strategy.set_queries(queries)
# Only Postgresql and MariaDb available at the moment # Only Postgresql and MariaDb available at the moment
def setup_db(self, index_config): def setup_db(self, index_config, reload_data=False):
setup.setupBoth(index_config) setup.reset_postgres(index_config, reload_data=reload_data)
def execute(self): def execute(self):
self.strategy.run() self.strategy.run()
+51 -11
View File
@@ -34,7 +34,15 @@ INDEX_CONFIGS = {
} }
def create_distribute_postgres(index_config): KNOWN_INDEX_NAMES = ("publ_pubid_idx", "auth_pubid_idx")
def load_data_postgres():
"""Drops and re-creates auth/publ in PostgreSQL and bulk-loads the TSV files.
Call this once at the start of a run, then use apply_indexes_postgres
between tests instead of reloading the whole dataset.
"""
_, connection = get_connection(maria=False) _, connection = get_connection(maria=False)
cursor = connection.cursor() cursor = connection.cursor()
@@ -42,8 +50,8 @@ def create_distribute_postgres(index_config):
start = time.time() start = time.time()
file_auth = open(f"{os.getenv("PATH_AUTH")}", "r", encoding="utf-8") file_auth = open(f"{os.getenv('PATH_AUTH')}", "r", encoding="utf-8")
file_publ = open(f"{os.getenv("PATH_PUBL")}", "r", encoding="utf-8") file_publ = open(f"{os.getenv('PATH_PUBL')}", "r", encoding="utf-8")
cursor.copy_from(file_auth, "auth", sep="\t", columns=("name", "pubid")) cursor.copy_from(file_auth, "auth", sep="\t", columns=("name", "pubid"))
cursor.copy_from( cursor.copy_from(
@@ -55,28 +63,60 @@ def create_distribute_postgres(index_config):
connection.commit() connection.commit()
end = time.time() end = time.time()
entries = 0
cursor.execute(COUNT_AUTH_ENTRIES_QUERY) cursor.execute(COUNT_AUTH_ENTRIES_QUERY)
entries += cursor.fetchall()[0][0] auth_entries = cursor.fetchall()[0][0]
print("Entries Auth: " + str(entries))
cursor.execute(COUNT_PUBL_ENTRIES_QUERY) cursor.execute(COUNT_PUBL_ENTRIES_QUERY)
publ_entries = cursor.fetchall()[0][0] publ_entries = cursor.fetchall()[0][0]
print("Entries Publ: " + str(publ_entries))
entries += publ_entries
print("Total Entries (Auth, Publ): " + str(entries)) print(f"Entries Auth: {auth_entries}")
print(f"Entries Publ: {publ_entries}")
print(f"Total Entries (Auth, Publ): {auth_entries + publ_entries}")
print(f"PostgreSQL Load Runtime: {end - start:.2f} seconds")
print("PostgreSQL Runtime:", end - start, "seconds") cursor.close()
connection.close()
def apply_indexes_postgres(index_config):
"""Drops all known indexes and applies the given index configuration.
Note: CLUSTER physically reorders the table. Dropping the index afterwards
does NOT undo that ordering. If a test needs the original physical order
after a previous 'cl-both' run, call load_data_postgres() again.
"""
_, connection = get_connection(maria=False)
cursor = connection.cursor()
for idx_name in KNOWN_INDEX_NAMES:
cursor.execute(f"DROP INDEX IF EXISTS {idx_name};")
for command in INDEX_CONFIGS[index_config]: for command in INDEX_CONFIGS[index_config]:
print("Applying:", command) print("Applying:", command)
cursor.execute(command) cursor.execute(command)
connection.commit()
connection.commit()
cursor.close() cursor.close()
connection.close() connection.close()
def reset_postgres(index_config, reload_data=False):
"""Bring PostgreSQL into the desired state for the next test.
Set reload_data=True when the previous test clustered the table and the
next test needs a non-clustered physical layout.
"""
if reload_data:
load_data_postgres()
apply_indexes_postgres(index_config)
def create_distribute_postgres(index_config):
"""Legacy: full reload + index setup in PostgreSQL. Kept for the CLI."""
load_data_postgres()
apply_indexes_postgres(index_config)
def create_distribute_maria(index_config): def create_distribute_maria(index_config):
_ ,connection = get_connection(maria=True) _ ,connection = get_connection(maria=True)
cursor = connection.cursor() cursor = connection.cursor()