commit 7a46ef37597dd347191dc4f4537251b5acb84ef6 Author: Sandro Fuetsch Titan Date: Tue May 26 19:23:46 2026 +0200 first commit diff --git a/.env b/.env new file mode 100644 index 0000000..99ee662 --- /dev/null +++ b/.env @@ -0,0 +1,10 @@ +MARIA_USER = "root" +MARIA_PASS = 1234 +MARIA_DB = "finetuning" + +POSTGRES_USER = "finetuner" +POSTGRES_PASS = 1234 +POSTGRES_DB = "finetuning" + +PATH_AUTH = "repositories/auth.tsv" +PATH_PUBL = "repositories/publ.tsv" \ No newline at end of file diff --git a/__pycache__/baseStrategy.cpython-313.pyc b/__pycache__/baseStrategy.cpython-313.pyc new file mode 100644 index 0000000..8123204 Binary files /dev/null and b/__pycache__/baseStrategy.cpython-313.pyc differ diff --git a/__pycache__/hashJoinStrategy.cpython-313.pyc b/__pycache__/hashJoinStrategy.cpython-313.pyc new file mode 100644 index 0000000..d3cf749 Binary files /dev/null and b/__pycache__/hashJoinStrategy.cpython-313.pyc differ diff --git a/__pycache__/manager.cpython-313.pyc b/__pycache__/manager.cpython-313.pyc new file mode 100644 index 0000000..6377f6b Binary files /dev/null and b/__pycache__/manager.cpython-313.pyc differ diff --git a/__pycache__/nestedInnerLoopStrategy.cpython-313.pyc b/__pycache__/nestedInnerLoopStrategy.cpython-313.pyc new file mode 100644 index 0000000..6c03bb3 Binary files /dev/null and b/__pycache__/nestedInnerLoopStrategy.cpython-313.pyc differ diff --git a/__pycache__/setup.cpython-313.pyc b/__pycache__/setup.cpython-313.pyc new file mode 100644 index 0000000..73a0df5 Binary files /dev/null and b/__pycache__/setup.cpython-313.pyc differ diff --git a/__pycache__/sortMergeStrategy.cpython-313.pyc b/__pycache__/sortMergeStrategy.cpython-313.pyc new file mode 100644 index 0000000..aef1d9a Binary files /dev/null and b/__pycache__/sortMergeStrategy.cpython-313.pyc differ diff --git a/assignment_04.pdf b/assignment_04.pdf new file mode 100644 index 0000000..5fc2aa8 Binary files /dev/null and b/assignment_04.pdf differ diff --git a/baseStrategy.py b/baseStrategy.py new file mode 100644 index 0000000..4aef0d9 --- /dev/null +++ b/baseStrategy.py @@ -0,0 +1,40 @@ +from setup import get_connection +import time +import psycopg2 +import mariadb + + + +class BaseStrategy: + + def __init__(self, conn: psycopg2.extensions.connection | mariadb.Connection, db_name, queries: dict[str, str]): + self.connection = conn + self.cursor = conn.cursor() + self.db_name = db_name + self.queries = queries + + def set_queries(self, queries): + self.queries = queries + + def run(self): + for strategy, query in self.queries: + print("Running: " + strategy + "\n") + start = time.time() + self.cursor.execute(query) + end = time.time() + self._print_result(strategy, end, start) + + + def _print_result(self, strategy, end, start): + duration = end - start + minutes= int(duration // 60) + seconds = duration % 60 + + print(f"---- RESULTS FROM {self.db_name} Strategy: {strategy} ----\n") + print(f"duration: {minutes}m {seconds:.2f}s") + _print_cursor(self.cursor) + +def _print_cursor(cursor): + for row in cursor.fetchall(): + print(f"{row} '\n'") + diff --git a/gruppenname_04.tex b/gruppenname_04.tex new file mode 100644 index 0000000..1874ba5 --- /dev/null +++ b/gruppenname_04.tex @@ -0,0 +1,189 @@ +\documentclass[11pt]{scrartcl} + +\usepackage[top=1.5cm]{geometry} +\usepackage{float} + +\setlength{\parindent}{0em} +\setlength{\parskip}{0.5em} + +\newcommand{\youranswerhere}{[Your answer goes here \ldots]} +\renewcommand{\thesubsection}{\arabic{subsection}} + +\title{ + \textbf{\large Assignment 5} \\ + Join Tuning \\ + {\large Database Tuning}} + +\author{ + Group Name (e.g. A1, B5, B3) \\ + \large Lastname1 Firstname1, StudentID1 \\ + \large Lastname2 Firstname2, StudentID2 \\ + \large Lastname3 Firstname3, StudentID3 +} + +\begin{document} + +\maketitle + +\subsection*{Experimental Setup} + +Describe your experimental setup in a few lines. + +\youranswerhere{} + +\subsection*{Join Strategies Proposed by System} + +\paragraph{Response times}\mbox{} + +\begin{table}[H] + \centering + \begin{tabular}{l|l|l} + Indexes & Join Strategy Q1 & Join Strategy Q2 \tabularnewline + \hline + no index & \ldots & \ldots \tabularnewline + unique non-clustering on \texttt{Publ.pubID} & \ldots & \ldots + \tabularnewline + clustering on \texttt{Publ.pubID} and \texttt{Auth.pubID} & \ldots & \ldots + \tabularnewline + \end{tabular} +\end{table} + +\paragraph{Discussion} + +Discuss your observations. Is the choice of the strategy expected? How does the system come to this choice? + +\youranswerhere{} + +\subsection*{Indexed Nested Loop Join} + +\paragraph{Response times}\mbox{} + +\begin{table}[H] + \centering + \begin{tabular}{l|r|r} + Indexes & Response time Q1 [ms] & Response time Q2 [ms] \tabularnewline + \hline + index on \texttt{Publ.pubID} & \ldots & \ldots \tabularnewline + index on \texttt{Auth.pubID} & \ldots & \ldots \tabularnewline + index on \texttt{Publ.pubID} and \texttt{Auth.pubID} & \ldots & \ldots + \tabularnewline + \end{tabular} +\end{table} + +\paragraph{Query plans}\mbox{} + +Index on \texttt{Publ.pubID} (Q1/Q2): + +{\small +\parskip0pt\begin{verbatim} +[Your query plans (index on Publ.pubID) go here ...] +\end{verbatim}} + +Index on \texttt{Auth.pubID} (Q1/Q2): + +{\small +\parskip0pt\begin{verbatim} +[Your query plans (index on Auth.pubID) go here ...] +\end{verbatim}} + +Index on \texttt{Auth.pubID} and \texttt{Auth.pubID} (Q1/Q2): + +{\small +\parskip0pt\begin{verbatim} +[Your query plans (index on Publ.pubID and Auth.pubID) go here ...] +\end{verbatim}} + +\paragraph{Discussion} + +Discuss your observations. Are the response times expected? Why (not)? + +\youranswerhere{} + +\subsection*{Sort-Merge Join} + +\paragraph{Response times}\mbox{} + +\begin{table}[H] + \centering + \begin{tabular}{l|r|r} + Indexes & Response time Q1 [ms] & Response time Q2 [ms] \tabularnewline + \hline + no index & \ldots & \ldots \tabularnewline + two non-clustering indexes & \ldots & \ldots \tabularnewline + two clustering indexes & \ldots & \ldots \tabularnewline + \end{tabular} +\end{table} + +\paragraph{Query plans}\mbox{} + +No index (Q1/Q2): + +{\small +\parskip0pt\begin{verbatim} +[Your query plans (no index) go here ...] +\end{verbatim}} + +Two non-clustering indexes (Q1/Q2): +{\small +\parskip0pt\begin{verbatim} +[Your query plans (two non-clustering indexes) go here ...] +\end{verbatim}} + +Two clustering indexes (Q1/Q2): +{\small +\parskip0pt\begin{verbatim} +[Your query plans (two clustering indexes) go here ...] +\end{verbatim}} + +\paragraph{Discussion} + +Discuss your observations. Are the response times expected? Why (not)? + +\youranswerhere{} + +\subsection*{Hash Join} + +\paragraph{Response times}\mbox{} + +\begin{table}[H] + \centering + \begin{tabular}{l|r|r} + Indexes & Response time Q1 [ms] & Response time [ms] Q2 \tabularnewline + \hline + no index & \ldots & \ldots \tabularnewline + \end{tabular} +\end{table} + +\paragraph{Query plans}\mbox{} + +No Index (Q1/Q2): + +{\small +\parskip0pt\begin{verbatim} +[Your query plans (no index) go here ...] +\end{verbatim}} + +\paragraph{Discussion} + +What do you think about the response time of the hash index vs.\ the response times of sort-merge and index nested loop join for each of the queries? Explain. + +\youranswerhere{} + +\subsection*{Time Spent on this Assignment} + +Time in hours per person: \textbf{XXX} + +\subsection*{References} + +\begin{table}[H] + \centering + \begin{tabular}{c} + \hline + \textbf{Important:} Reference your information sources! \tabularnewline + Remove this section if you use footnotes to reference your information + sources. \tabularnewline + \hline + \end{tabular} +\end{table} + +\end{document} diff --git a/hashJoinStrategy.py b/hashJoinStrategy.py new file mode 100644 index 0000000..9e64158 --- /dev/null +++ b/hashJoinStrategy.py @@ -0,0 +1,18 @@ +from repositories.hash_join_resolver import resolve +from baseStrategy import BaseStrategy + +class HashJoinStrategy(BaseStrategy): + def __init__(self, conn, db_name, all_queries): + super().__init__(conn, db_name, all_queries) + + def run(self): + if self.db_name not in resolve: + raise Exception("Hash Join not supported for " + self.db_name) + + for disable in resolve[self.db_name]["disable"]: + self.cursor.execute(disable) + + super().run() + + for reset in resolve[self.db_name]["reset"]: + self.cursor.execute(reset) \ No newline at end of file diff --git a/main.py b/main.py new file mode 100644 index 0000000..1c81b0f --- /dev/null +++ b/main.py @@ -0,0 +1,49 @@ +from manager import Manager +from baseStrategy import BaseStrategy +from nestedInnerLoopStrategy import NestedInnerLoopStrategy +from sortMergeStrategy import SortMergeStrategy +from hashJoinStrategy import HashJoinStrategy +from setup import get_connection +from repositories.all_queries import queries_postgres +if __name__ == '__main__': + join_manager: Manager | None = None + + + db_post, conn_postsql = get_connection(maria=False) + queries_explict_no_index_postgresql = queries_postgres["no_index"] + + + #Test on Postgresql + #Aufgabe 1 + join_manager = Manager(BaseStrategy(conn_postsql, db_post, queries_postgres["with_index"])) + #join_manager.setup_db("no-index") + #join_manager.setQueries(queries_ignore_index) 3 Billionen Einträge durch kreuzprodukt > 10min + #join_manager.execute() + join_manager.setup_db("unique-publ") + join_manager.execute() + join_manager.setup_db("cl-both") + join_manager.execute() + #Aufgabe 2 + join_manager.setStrategy(NestedInnerLoopStrategy(conn_postsql, db_post, queries_postgres["with_index"])) + join_manager.setup_db("nc-publ") + join_manager.execute() + join_manager.setup_db("nc-auth") + join_manager.execute() + join_manager.setup_db("nc-both") + join_manager.execute() + #Aufgabe 3 + join_manager.setStrategy(SortMergeStrategy(conn_postsql, db_post, queries_postgres["no_index"])) + join_manager.setup_db("no-index") + join_manager.execute() + join_manager.setQueries(queries_postgres["with_index"]) + join_manager.setup_db("nc-both") + join_manager.execute() + join_manager.setup_db("cl-both") + #Aufgabe 4 + join_manager.setStrategy(HashJoinStrategy(conn_postsql, db_post, queries_postgres["no_index"])) + join_manager.setup_db("no-index") + join_manager.execute() + + + + \ No newline at end of file diff --git a/manager.py b/manager.py new file mode 100644 index 0000000..32bf788 --- /dev/null +++ b/manager.py @@ -0,0 +1,20 @@ +from baseStrategy import BaseStrategy +import setup + +class Manager: + def __init__(self, strategy: BaseStrategy, ): + self.strategy = strategy + + def setStrategy(self, strategy: BaseStrategy): + self.strategy = strategy + + def setQueries(self, queries: dict[str, str]): + self.strategy.set_queries(queries) + + # Only Postgresql and MariaDb available at the moment + def setup_db(self, index_config): + setup.setupBoth(index_config) + + def execute(self): + self.strategy.run() + \ No newline at end of file diff --git a/nestedInnerLoopStrategy.py b/nestedInnerLoopStrategy.py new file mode 100644 index 0000000..ffae320 --- /dev/null +++ b/nestedInnerLoopStrategy.py @@ -0,0 +1,18 @@ +from repositories.nested_loop_resolver import resolve +from baseStrategy import BaseStrategy + +class NestedInnerLoopStrategy(BaseStrategy): + def __init__(self, conn, db_name, all_queries): + super().__init__(conn, db_name, all_queries) + + def run(self): + if self.db_name not in resolve: + raise Exception("No Inner Loop Resolver Found") + + for disable in resolve[self.db_name]["disable"]: + self.cursor.execute(disable) + + super().run() + + for reset in resolve[self.db_name]["reset"]: + self.cursor.execute(reset) \ No newline at end of file diff --git a/repositories/__pycache__/all_queries.cpython-313.pyc b/repositories/__pycache__/all_queries.cpython-313.pyc new file mode 100644 index 0000000..dc0d7e9 Binary files /dev/null and b/repositories/__pycache__/all_queries.cpython-313.pyc differ diff --git a/repositories/__pycache__/hash_join_resolver.cpython-313.pyc b/repositories/__pycache__/hash_join_resolver.cpython-313.pyc new file mode 100644 index 0000000..4541764 Binary files /dev/null and b/repositories/__pycache__/hash_join_resolver.cpython-313.pyc differ diff --git a/repositories/__pycache__/nested_loop_resolver.cpython-313.pyc b/repositories/__pycache__/nested_loop_resolver.cpython-313.pyc new file mode 100644 index 0000000..86c68bb Binary files /dev/null and b/repositories/__pycache__/nested_loop_resolver.cpython-313.pyc differ diff --git a/repositories/__pycache__/sort_merge_resolver.cpython-313.pyc b/repositories/__pycache__/sort_merge_resolver.cpython-313.pyc new file mode 100644 index 0000000..4943f92 Binary files /dev/null and b/repositories/__pycache__/sort_merge_resolver.cpython-313.pyc differ diff --git a/repositories/all_queries.py b/repositories/all_queries.py new file mode 100644 index 0000000..067f9ab --- /dev/null +++ b/repositories/all_queries.py @@ -0,0 +1,53 @@ +queries_maria = { + "no_index": { + "query_1": """ + ANALYZE SELECT name, title + FROM Auth IGNORE INDEX (idx_auth_pubid), Publ IGNORE INDEX(idx_publ_pubid) + WHERE Auth.pubID = Publ.pubID; + """, + "query_2": """ + ANALYZE SELECT title + FROM Auth IGNORE INDEX (idx_auth_pubid), Publ IGNORE INDEX(idx_publ_pubid) + WHERE Auth.pubID = Publ.pubID AND Auth.name = 'Divesh Srivastava'; + """, + }, + "with_index": { + "query_1": """ + ANALYZE SELECT name, title + FROM Auth, Publ + WHERE Auth.pubID = Publ.pubID; + """, + "query_2": """ + ANALYZE SELECT title + FROM Auth, Publ + WHERE Auth.pubID = Publ.pubID AND Auth.name = 'Divesh Srivastava'; + """, + }, +} + +queries_postgres = { + "no_index": { + "query_1": """ + EXPLAIN ANALYZE SELECT name, title + FROM Auth, Publ + WHERE Auth.pubID = Publ.pubID; + """, + "query_2": """ + EXPLAIN ANALYZE SELECT title + FROM Auth, Publ + WHERE Auth.pubID = Publ.pubID AND Auth.name = 'Divesh Srivastava'; + """, + }, + "with_index": { + "query_1": """ + EXPLAIN ANALYZE SELECT name, title + FROM Auth, Publ + WHERE Auth.pubID = Publ.pubID; + """, + "query_2": """ + EXPLAIN ANALYZE SELECT title + FROM Auth, Publ + WHERE Auth.pubID = Publ.pubID AND Auth.name = 'Divesh Srivastava'; + """, + }, +} \ No newline at end of file diff --git a/repositories/hash_join_resolver.py b/repositories/hash_join_resolver.py new file mode 100644 index 0000000..a2828a6 --- /dev/null +++ b/repositories/hash_join_resolver.py @@ -0,0 +1,10 @@ +resolve = { + "postgresql": { + "disable": ["SET enable_nestloop = off", "SET enable_mergejoin = off"], + "reset": ["RESET enable_nestloop", "RESET enable_mergejoin"], + }, + "mariadb": { + "disable": ["SET optimizer_switch='join_cache_hashed=on'", "SET join_cache_level=8"], + "reset": ["SET optimizer_switch=DEFAULT", "SET join_cache_level=DEFAULT"], + }, +} \ No newline at end of file diff --git a/repositories/nested_loop_resolver.py b/repositories/nested_loop_resolver.py new file mode 100644 index 0000000..168ba23 --- /dev/null +++ b/repositories/nested_loop_resolver.py @@ -0,0 +1,11 @@ +# nestedLoopResolver.py +resolve = { + "postgresql": { + "disable": ["SET enable_hashjoin = off", "SET enable_mergejoin = off"], + "reset": ["RESET enable_hashjoin", "RESET enable_mergejoin"], + }, + "mariadb": { + "disable": ["SET optimizer_switch='block_nested_loop=off'"], + "reset": ["SET optimizer_switch=DEFAULT"], + }, +} \ No newline at end of file diff --git a/repositories/sort_merge_resolver.py b/repositories/sort_merge_resolver.py new file mode 100644 index 0000000..d26d085 --- /dev/null +++ b/repositories/sort_merge_resolver.py @@ -0,0 +1,7 @@ +# sortMergeResolver.py +resolve = { + "postgresql": { + "disable": ["SET enable_hashjoin = off", "SET enable_nestloop = off"], + "reset": ["RESET enable_hashjoin", "RESET enable_nestloop"], + }, +} \ No newline at end of file diff --git a/setup.py b/setup.py new file mode 100644 index 0000000..0a2ee16 --- /dev/null +++ b/setup.py @@ -0,0 +1,193 @@ +import mariadb +import psycopg2 +import time +import os +import argparse +from dotenv import load_dotenv + +load_dotenv() + +COUNT_AUTH_ENTRIES_QUERY = "SELECT COUNT(*) FROM auth" +COUNT_PUBL_ENTRIES_QUERY = "SELECT COUNT(*) FROM publ" + +INDEX_CONFIGS = { + "no-index": [], + "unique-publ": [ + "CREATE UNIQUE INDEX publ_pubid_idx ON publ(pubid);", + ], + "nc-publ": [ + "CREATE INDEX publ_pubid_idx ON publ(pubid);", + ], + "nc-auth": [ + "CREATE INDEX auth_pubid_idx ON auth(pubid);", + ], + "nc-both": [ + "CREATE INDEX publ_pubid_idx ON publ(pubid);", + "CREATE INDEX auth_pubid_idx ON auth(pubid);", + ], + "cl-both": [ + "CREATE INDEX publ_pubid_idx ON publ(pubid);", + "CLUSTER publ USING publ_pubid_idx;", + "CREATE INDEX auth_pubid_idx ON auth(pubid);", + "CLUSTER auth USING auth_pubid_idx;", + ], +} + + +def create_distribute_postgres(index_config): + _, connection = get_connection(maria=False) + cursor = connection.cursor() + + _create_tables_force(cursor) + + start = time.time() + + file_auth = open(f"{os.getenv("PATH_AUTH")}", "r", encoding="utf-8") + file_publ = open(f"{os.getenv("PATH_PUBL")}", "r", encoding="utf-8") + + cursor.copy_from(file_auth, "auth", sep="\t", columns=("name", "pubid")) + cursor.copy_from( + file_publ, + "publ", + sep="\t", + columns=("pubid", "type", "title", "booktitle", "year", "publisher"), + ) + + connection.commit() + end = time.time() + entries = 0 + cursor.execute(COUNT_AUTH_ENTRIES_QUERY) + entries += cursor.fetchall()[0][0] + print("Entries Auth: " + str(entries)) + cursor.execute(COUNT_PUBL_ENTRIES_QUERY) + publ_entries = cursor.fetchall()[0][0] + print("Entries Publ: " + str(publ_entries)) + entries += publ_entries + + print("Total Entries (Auth, Publ): " + str(entries)) + + print("PostgreSQL Runtime:", end - start, "seconds") + + for command in INDEX_CONFIGS[index_config]: + print("Applying:", command) + cursor.execute(command) + connection.commit() + + cursor.close() + connection.close() + + +def create_distribute_maria(index_config): + _ ,connection = get_connection(maria=True) + cursor = connection.cursor() + _create_tables_force(cursor) + + start = time.time() + + cursor.execute(f""" + LOAD DATA LOCAL INFILE '{os.getenv("PATH_AUTH")}' + INTO TABLE auth + FIELDS TERMINATED BY '\\t' + LINES TERMINATED BY '\\n' + (name, pubid) + """) + + cursor.execute(f""" + LOAD DATA LOCAL INFILE '{os.getenv("PATH_PUBL")}' + INTO TABLE publ + FIELDS TERMINATED BY '\\t' + LINES TERMINATED BY '\\n' + (pubid, type, title, booktitle, year, publisher) + """) + + connection.commit() + end = time.time() + + print("MariaDB Runtime:", end - start, "seconds") + + for command in INDEX_CONFIGS[index_config]: + if command.startswith("CLUSTER"): + print("Skipping (not supported in MariaDB):", command) + continue + print("Applying:", command) + cursor.execute(command) + connection.commit() + + cursor.execute(COUNT_AUTH_ENTRIES_QUERY) + entries = cursor.fetchall()[0][0] + print("Entries Auth: " + str(entries)) + cursor.execute(COUNT_PUBL_ENTRIES_QUERY) + publ_entries = cursor.fetchall()[0][0] + print("Entries Publ: " + str(publ_entries)) + entries += publ_entries + + print("Total Entries (Auth, Publ): " + str(entries)) + + cursor.close() + connection.close() + + +def get_connection(maria: bool) -> tuple[str, mariadb.Connection | psycopg2.extensions.connection]: + connection = None + db_name = "None" + if not maria: + connection = psycopg2.connect( + dbname=os.getenv("POSTGRES_DB"), + user=os.getenv("POSTGRES_USER"), + password=os.getenv("POSTGRES_PASS"), + host="localhost", + port="5432", + ) + db_name = "postgresql" + + else: + connection = mariadb.connect( + user=os.getenv("MARIA_USER"), + password=os.getenv("MARIA_PASS"), + host="localhost", + port=3306, + database=os.getenv("MARIA_DB"), + local_infile=True, + ) + db_name = "mariadb" + + return (db_name, connection) + + +def _create_tables_force(cursor): # type: ignore + + cursor.execute("DROP TABLE IF EXISTS auth") + cursor.execute("DROP TABLE IF EXISTS publ") + + cursor.execute("""CREATE TABLE auth ( + name VARCHAR(49), + pubid VARCHAR(129) + + )""") + + cursor.execute(""" + CREATE TABLE publ ( + pubid VARCHAR(129), + type VARCHAR(13), + title VARCHAR(700), + booktitle VARCHAR(132), + year VARCHAR(4), + publisher VARCHAR(196) + ) + """) + + +def setupBoth(index_config): + create_distribute_postgres(index_config) + create_distribute_maria(index_config) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument( + "index", choices=INDEX_CONFIGS.keys(), help="Index configuration to apply" + ) + args = parser.parse_args() + + print(f"Index: {args.index}") + setupBoth(args.index) diff --git a/sortMergeStrategy.py b/sortMergeStrategy.py new file mode 100644 index 0000000..c04efc7 --- /dev/null +++ b/sortMergeStrategy.py @@ -0,0 +1,18 @@ +from repositories.sort_merge_resolver import resolve +from baseStrategy import BaseStrategy + +class SortMergeStrategy(BaseStrategy): + def __init__(self, conn, db_name, all_queries): + super().__init__(conn, db_name, all_queries) + + def run(self): + if self.db_name not in resolve: + raise Exception("Sort-Merge Join not supported for " + self.db_name) + + for disable in resolve[self.db_name]["disable"]: + self.cursor.execute(disable) + + super().run() + + for reset in resolve[self.db_name]["reset"]: + self.cursor.execute(reset) \ No newline at end of file