From d8a906e2d46a14d76a227500637f391181325c88 Mon Sep 17 00:00:00 2001 From: Sandro Fuetsch Titan Date: Tue, 26 May 2026 19:50:07 +0200 Subject: [PATCH] Load DBLP data once instead of reloading per index config The Manager previously called setupBoth() on every test step, which dropped both PostgreSQL and MariaDB tables and re-imported the full DBLP TSVs (~9 reloads across 2 DBs in a single run, even though only PostgreSQL was used). Now data is loaded once via load_data_postgres() and only the indexes are swapped between tests via apply_indexes_postgres(). A reset_postgres(config, reload_data=False) helper lets callers force a full reload when needed -- used after 'cl-both' tests, since CLUSTER physically reorders the table and dropping the index alone does not restore the original layout. Also fixes two bugs found while reading the code: - baseStrategy.run iterated 'self.queries' instead of '.items()', which would crash on the first unpack. - main.py was missing an execute() call after the Aufgabe 3 'cl-both' setup, silently skipping that test. Co-Authored-By: Claude Opus 4.7 --- baseStrategy.py | 2 +- main.py | 96 ++++++++++++++++++++++++------------------------- manager.py | 4 +-- setup.py | 62 ++++++++++++++++++++++++++------ 4 files changed, 101 insertions(+), 63 deletions(-) diff --git a/baseStrategy.py b/baseStrategy.py index 4aef0d9..4ed0e8d 100644 --- a/baseStrategy.py +++ b/baseStrategy.py @@ -17,7 +17,7 @@ class BaseStrategy: self.queries = queries def run(self): - for strategy, query in self.queries: + for strategy, query in self.queries.items(): print("Running: " + strategy + "\n") start = time.time() self.cursor.execute(query) diff --git a/main.py b/main.py index 1c81b0f..8de1681 100644 --- a/main.py +++ b/main.py @@ -1,49 +1,47 @@ -from manager import Manager -from baseStrategy import BaseStrategy -from nestedInnerLoopStrategy import NestedInnerLoopStrategy -from sortMergeStrategy import SortMergeStrategy -from hashJoinStrategy import HashJoinStrategy -from setup import get_connection -from repositories.all_queries import queries_postgres -if __name__ == '__main__': - join_manager: Manager | None = None - - - db_post, conn_postsql = get_connection(maria=False) - queries_explict_no_index_postgresql = queries_postgres["no_index"] - - - #Test on Postgresql - #Aufgabe 1 - join_manager = Manager(BaseStrategy(conn_postsql, db_post, queries_postgres["with_index"])) - #join_manager.setup_db("no-index") - #join_manager.setQueries(queries_ignore_index) 3 Billionen Einträge durch kreuzprodukt > 10min - #join_manager.execute() - join_manager.setup_db("unique-publ") - join_manager.execute() - join_manager.setup_db("cl-both") - join_manager.execute() - #Aufgabe 2 - join_manager.setStrategy(NestedInnerLoopStrategy(conn_postsql, db_post, queries_postgres["with_index"])) - join_manager.setup_db("nc-publ") - join_manager.execute() - join_manager.setup_db("nc-auth") - join_manager.execute() - join_manager.setup_db("nc-both") - join_manager.execute() - #Aufgabe 3 - join_manager.setStrategy(SortMergeStrategy(conn_postsql, db_post, queries_postgres["no_index"])) - join_manager.setup_db("no-index") - join_manager.execute() - join_manager.setQueries(queries_postgres["with_index"]) - join_manager.setup_db("nc-both") - join_manager.execute() - join_manager.setup_db("cl-both") - #Aufgabe 4 - join_manager.setStrategy(HashJoinStrategy(conn_postsql, db_post, queries_postgres["no_index"])) - join_manager.setup_db("no-index") - join_manager.execute() - - - - \ No newline at end of file +from manager import Manager +from baseStrategy import BaseStrategy +from nestedInnerLoopStrategy import NestedInnerLoopStrategy +from sortMergeStrategy import SortMergeStrategy +from hashJoinStrategy import HashJoinStrategy +from setup import get_connection, load_data_postgres +from repositories.all_queries import queries_postgres + +if __name__ == '__main__': + db_post, conn_postsql = get_connection(maria=False) + + # Daten einmalig laden; danach werden zwischen den Tests nur die Indexe getauscht. + load_data_postgres() + + # Aufgabe 1 + join_manager = Manager(BaseStrategy(conn_postsql, db_post, queries_postgres["with_index"])) + # join_manager.setup_db("no-index") + # join_manager.execute() # 3 Mrd. Tupel via Kreuzprodukt -> > 10 min + join_manager.setup_db("unique-publ") + join_manager.execute() + join_manager.setup_db("cl-both") + join_manager.execute() + + # Aufgabe 2 — vorheriger Lauf war 'cl-both' (CLUSTER hat Tabelle physisch sortiert); + # reload_data=True stellt die ursprüngliche Ladereihenfolge wieder her. + join_manager.setStrategy(NestedInnerLoopStrategy(conn_postsql, db_post, queries_postgres["with_index"])) + join_manager.setup_db("nc-publ", reload_data=True) + join_manager.execute() + join_manager.setup_db("nc-auth") + join_manager.execute() + join_manager.setup_db("nc-both") + join_manager.execute() + + # Aufgabe 3 — Tabelle ist noch im Original-Layout, kein Reload nötig. + join_manager.setStrategy(SortMergeStrategy(conn_postsql, db_post, queries_postgres["no_index"])) + join_manager.setup_db("no-index") + join_manager.execute() + join_manager.setQueries(queries_postgres["with_index"]) + join_manager.setup_db("nc-both") + join_manager.execute() + join_manager.setup_db("cl-both") + join_manager.execute() + + # Aufgabe 4 — wieder vom CLUSTER-Zustand wegkommen. + join_manager.setStrategy(HashJoinStrategy(conn_postsql, db_post, queries_postgres["no_index"])) + join_manager.setup_db("no-index", reload_data=True) + join_manager.execute() diff --git a/manager.py b/manager.py index 32bf788..c066537 100644 --- a/manager.py +++ b/manager.py @@ -12,8 +12,8 @@ class Manager: self.strategy.set_queries(queries) # Only Postgresql and MariaDb available at the moment - def setup_db(self, index_config): - setup.setupBoth(index_config) + def setup_db(self, index_config, reload_data=False): + setup.reset_postgres(index_config, reload_data=reload_data) def execute(self): self.strategy.run() diff --git a/setup.py b/setup.py index 0a2ee16..a47bffa 100644 --- a/setup.py +++ b/setup.py @@ -34,7 +34,15 @@ INDEX_CONFIGS = { } -def create_distribute_postgres(index_config): +KNOWN_INDEX_NAMES = ("publ_pubid_idx", "auth_pubid_idx") + + +def load_data_postgres(): + """Drops and re-creates auth/publ in PostgreSQL and bulk-loads the TSV files. + + Call this once at the start of a run, then use apply_indexes_postgres + between tests instead of reloading the whole dataset. + """ _, connection = get_connection(maria=False) cursor = connection.cursor() @@ -42,8 +50,8 @@ def create_distribute_postgres(index_config): start = time.time() - file_auth = open(f"{os.getenv("PATH_AUTH")}", "r", encoding="utf-8") - file_publ = open(f"{os.getenv("PATH_PUBL")}", "r", encoding="utf-8") + file_auth = open(f"{os.getenv('PATH_AUTH')}", "r", encoding="utf-8") + file_publ = open(f"{os.getenv('PATH_PUBL')}", "r", encoding="utf-8") cursor.copy_from(file_auth, "auth", sep="\t", columns=("name", "pubid")) cursor.copy_from( @@ -55,28 +63,60 @@ def create_distribute_postgres(index_config): connection.commit() end = time.time() - entries = 0 + cursor.execute(COUNT_AUTH_ENTRIES_QUERY) - entries += cursor.fetchall()[0][0] - print("Entries Auth: " + str(entries)) + auth_entries = cursor.fetchall()[0][0] cursor.execute(COUNT_PUBL_ENTRIES_QUERY) publ_entries = cursor.fetchall()[0][0] - print("Entries Publ: " + str(publ_entries)) - entries += publ_entries - print("Total Entries (Auth, Publ): " + str(entries)) + print(f"Entries Auth: {auth_entries}") + print(f"Entries Publ: {publ_entries}") + print(f"Total Entries (Auth, Publ): {auth_entries + publ_entries}") + print(f"PostgreSQL Load Runtime: {end - start:.2f} seconds") - print("PostgreSQL Runtime:", end - start, "seconds") + cursor.close() + connection.close() + + +def apply_indexes_postgres(index_config): + """Drops all known indexes and applies the given index configuration. + + Note: CLUSTER physically reorders the table. Dropping the index afterwards + does NOT undo that ordering. If a test needs the original physical order + after a previous 'cl-both' run, call load_data_postgres() again. + """ + _, connection = get_connection(maria=False) + cursor = connection.cursor() + + for idx_name in KNOWN_INDEX_NAMES: + cursor.execute(f"DROP INDEX IF EXISTS {idx_name};") for command in INDEX_CONFIGS[index_config]: print("Applying:", command) cursor.execute(command) - connection.commit() + connection.commit() cursor.close() connection.close() +def reset_postgres(index_config, reload_data=False): + """Bring PostgreSQL into the desired state for the next test. + + Set reload_data=True when the previous test clustered the table and the + next test needs a non-clustered physical layout. + """ + if reload_data: + load_data_postgres() + apply_indexes_postgres(index_config) + + +def create_distribute_postgres(index_config): + """Legacy: full reload + index setup in PostgreSQL. Kept for the CLI.""" + load_data_postgres() + apply_indexes_postgres(index_config) + + def create_distribute_maria(index_config): _ ,connection = get_connection(maria=True) cursor = connection.cursor()