mirror of
https://github.com/sandronator/finetuning_aufgabe4.git
synced 2026-09-04 08:36:09 +02:00
Two related improvements to make the test runs both usable and
diagnosable:
1. Avoid lock waits / "deadlock" hangs during CLUSTER
The Manager connection ran EXPLAIN ANALYZE without committing, so
psycopg2's default autocommit=False left an open transaction
holding AccessShareLock on publ and auth. The next setup_db then
blocked on DROP INDEX / CLUSTER (both need AccessExclusiveLock),
visible as a hang -- typically at the CLUSTER step. Setting
conn_postsql.autocommit = True releases the locks immediately after
each SELECT, which is safe here because all Manager queries are
read-only.
2. Make the run output identify which test is running
The "Running: ..." line previously printed only the query key
(query_1/query_2), shadowing 'strategy' with the dict key and never
surfacing which join strategy or index config was active. Output
now reads e.g. "[Aufgabe 3b | SortMergeStrategy | idx=nc-both]" by:
- using self.__class__.__name__ so subclasses surface correctly
- passing the current index_config from Manager into the strategy
- threading an optional 'aufgabe' label through Manager.execute()
and into each strategy's run() (including the Nested/SortMerge/
Hash wrappers that override run()).
main.py now tags each execute() call with its assignment number.
Co-Authored-By: Claude Opus 4.7 <[email protected]>
51 lines
2.4 KiB
Python
51 lines
2.4 KiB
Python
from manager import Manager
|
|
from baseStrategy import BaseStrategy
|
|
from nestedInnerLoopStrategy import NestedInnerLoopStrategy
|
|
from sortMergeStrategy import SortMergeStrategy
|
|
from hashJoinStrategy import HashJoinStrategy
|
|
from setup import get_connection, load_data_postgres
|
|
from repositories.all_queries import queries_postgres
|
|
|
|
if __name__ == '__main__':
|
|
db_post, conn_postsql = get_connection(maria=False)
|
|
# Autocommit verhindert, dass EXPLAIN ANALYZE Locks (AccessShareLock auf publ/auth)
|
|
# bis zum naechsten Commit haelt -- sonst blockiert das spaeter DROP INDEX/CLUSTER.
|
|
conn_postsql.autocommit = True
|
|
|
|
# Daten einmalig laden; danach werden zwischen den Tests nur die Indexe getauscht.
|
|
load_data_postgres()
|
|
|
|
# Aufgabe 1
|
|
join_manager = Manager(BaseStrategy(conn_postsql, db_post, queries_postgres["with_index"]))
|
|
# join_manager.setup_db("no-index")
|
|
# join_manager.execute(aufgabe="Aufgabe 1a") # 3 Mrd. Tupel via Kreuzprodukt -> > 10 min
|
|
join_manager.setup_db("unique-publ")
|
|
join_manager.execute(aufgabe="Aufgabe 1b")
|
|
join_manager.setup_db("cl-both")
|
|
join_manager.execute(aufgabe="Aufgabe 1c")
|
|
|
|
# Aufgabe 2 — vorheriger Lauf war 'cl-both' (CLUSTER hat Tabelle physisch sortiert);
|
|
# reload_data=True stellt die ursprüngliche Ladereihenfolge wieder her.
|
|
join_manager.setStrategy(NestedInnerLoopStrategy(conn_postsql, db_post, queries_postgres["with_index"]))
|
|
join_manager.setup_db("nc-publ", reload_data=True)
|
|
join_manager.execute(aufgabe="Aufgabe 2a")
|
|
join_manager.setup_db("nc-auth")
|
|
join_manager.execute(aufgabe="Aufgabe 2b")
|
|
join_manager.setup_db("nc-both")
|
|
join_manager.execute(aufgabe="Aufgabe 2c")
|
|
|
|
# Aufgabe 3 — Tabelle ist noch im Original-Layout, kein Reload nötig.
|
|
join_manager.setStrategy(SortMergeStrategy(conn_postsql, db_post, queries_postgres["no_index"]))
|
|
join_manager.setup_db("no-index")
|
|
join_manager.execute(aufgabe="Aufgabe 3a")
|
|
join_manager.setQueries(queries_postgres["with_index"])
|
|
join_manager.setup_db("nc-both")
|
|
join_manager.execute(aufgabe="Aufgabe 3b")
|
|
join_manager.setup_db("cl-both")
|
|
join_manager.execute(aufgabe="Aufgabe 3c")
|
|
|
|
# Aufgabe 4 — wieder vom CLUSTER-Zustand wegkommen.
|
|
join_manager.setStrategy(HashJoinStrategy(conn_postsql, db_post, queries_postgres["no_index"]))
|
|
join_manager.setup_db("no-index", reload_data=True)
|
|
join_manager.execute(aufgabe="Aufgabe 4")
|