DavMelchi commited on
Commit
a381730
·
1 Parent(s): f53a89e

Reuse parsed database results across generators

Browse files
queries/process_all_db.py CHANGED
@@ -8,12 +8,10 @@ from queries.process_nice_db import process_data_for_nice
8
  from queries.process_site_db import site_db
9
  from queries.process_wcdma import process_wcdma_data, wcdma_analaysis
10
  from utils.convert_to_excel import convert_database_dfs, convert_dfs
11
- from utils.dump_excel import clear_dump_excel_cache
12
  from utils.utils_vars import UtilsVars
13
 
14
 
15
  def clear_all_dbs():
16
- clear_dump_excel_cache()
17
  UtilsVars.all_db_dfs.clear()
18
  UtilsVars.all_db_dfs_names.clear()
19
  UtilsVars.gsm_dfs.clear()
 
8
  from queries.process_site_db import site_db
9
  from queries.process_wcdma import process_wcdma_data, wcdma_analaysis
10
  from utils.convert_to_excel import convert_database_dfs, convert_dfs
 
11
  from utils.utils_vars import UtilsVars
12
 
13
 
14
  def clear_all_dbs():
 
15
  UtilsVars.all_db_dfs.clear()
16
  UtilsVars.all_db_dfs_names.clear()
17
  UtilsVars.gsm_dfs.clear()
queries/process_gsm.py CHANGED
@@ -7,6 +7,7 @@ from utils.config_band import bcf_band, config_band
7
  from utils.convert_to_excel import convert_dfs, save_dataframe
8
  from utils.dump_excel import read_dump_excel
9
  from utils.kml_creator import generate_kml_from_df
 
10
  from utils.utils_vars import (
11
  GsmAnalysisData,
12
  UtilsVars,
@@ -93,7 +94,7 @@ def compare_trx_tch_versus_mal(tch1, tch2):
93
  return set1 == set2
94
 
95
 
96
- def process_gsm_data(
97
  file_path: str,
98
  df_trx: pd.DataFrame | None = None,
99
  df_mal: pd.DataFrame | None = None,
@@ -201,13 +202,44 @@ def process_gsm_data(
201
  return df_2g
202
 
203
 
204
- def combined_gsm_database(file_path: str):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
205
  df_bts = process_small_bts_data(file_path)
206
  trx_df = process_trx_with_bts_name(file_path, df_bts=df_bts)
207
  trx_summary_df = process_trx_data(file_path, trx_bts_name=trx_df)
208
  mal_summary_df = process_mal_data(file_path)
209
  gsm_df = process_gsm_data(file_path, df_trx=trx_summary_df, df_mal=mal_summary_df)
 
210
  mal_df = process_mal_with_bts_name(file_path, mal_df=mal_summary_df, df_bts=df_bts)
 
 
 
 
 
 
 
 
 
211
 
212
  UtilsVars.all_db_dfs.extend([gsm_df, mal_df, trx_df])
213
  UtilsVars.gsm_dfs.extend([gsm_df, mal_df, trx_df])
 
7
  from utils.convert_to_excel import convert_dfs, save_dataframe
8
  from utils.dump_excel import read_dump_excel
9
  from utils.kml_creator import generate_kml_from_df
10
+ from utils.processing_cache import get_or_build_processed, remember_processed
11
  from utils.utils_vars import (
12
  GsmAnalysisData,
13
  UtilsVars,
 
94
  return set1 == set2
95
 
96
 
97
+ def _build_gsm_data(
98
  file_path: str,
99
  df_trx: pd.DataFrame | None = None,
100
  df_mal: pd.DataFrame | None = None,
 
202
  return df_2g
203
 
204
 
205
+ def process_gsm_data(
206
+ file_path: str,
207
+ df_trx: pd.DataFrame | None = None,
208
+ df_mal: pd.DataFrame | None = None,
209
+ ) -> pd.DataFrame:
210
+ """
211
+ Process GSM data.
212
+
213
+ Provided TRX/MAL dataframes are used directly by the All DB path to avoid
214
+ recomputing intermediate sheets. Plain calls are cached per dump/config.
215
+ """
216
+ if df_trx is not None or df_mal is not None:
217
+ return _build_gsm_data(file_path, df_trx=df_trx, df_mal=df_mal)
218
+
219
+ return get_or_build_processed(
220
+ file_path,
221
+ "gsm_data",
222
+ lambda: _build_gsm_data(file_path),
223
+ )
224
+
225
+
226
+ def _build_combined_gsm_database(file_path: str) -> list[pd.DataFrame]:
227
  df_bts = process_small_bts_data(file_path)
228
  trx_df = process_trx_with_bts_name(file_path, df_bts=df_bts)
229
  trx_summary_df = process_trx_data(file_path, trx_bts_name=trx_df)
230
  mal_summary_df = process_mal_data(file_path)
231
  gsm_df = process_gsm_data(file_path, df_trx=trx_summary_df, df_mal=mal_summary_df)
232
+ remember_processed(file_path, "gsm_data", gsm_df)
233
  mal_df = process_mal_with_bts_name(file_path, mal_df=mal_summary_df, df_bts=df_bts)
234
+ return [gsm_df, mal_df, trx_df]
235
+
236
+
237
+ def combined_gsm_database(file_path: str):
238
+ gsm_df, mal_df, trx_df = get_or_build_processed(
239
+ file_path,
240
+ "gsm_bundle",
241
+ lambda: _build_combined_gsm_database(file_path),
242
+ )
243
 
244
  UtilsVars.all_db_dfs.extend([gsm_df, mal_df, trx_df])
245
  UtilsVars.gsm_dfs.extend([gsm_df, mal_df, trx_df])
queries/process_invunit.py CHANGED
@@ -3,6 +3,7 @@ import pandas as pd
3
  from utils.convert_to_excel import convert_invunit_dfs, save_dataframe
4
  from utils.dump_excel import read_dump_excel
5
  from utils.extract_code import extract_code_from_mrbts
 
6
  from utils.utils_vars import UtilsVars
7
 
8
  RF_UNIT = [
@@ -112,7 +113,7 @@ def create_invunit_summary(df: pd.DataFrame) -> pd.DataFrame:
112
  return df
113
 
114
 
115
- def build_invunit_number_dataframe(file_path: str) -> pd.DataFrame:
116
  """
117
  Build detailed INVUNIT_NUMBER dataframe from dump INVUNIT sheet.
118
  """
@@ -189,6 +190,17 @@ def build_invunit_number_dataframe(file_path: str) -> pd.DataFrame:
189
  return df_invunit_number
190
 
191
 
 
 
 
 
 
 
 
 
 
 
 
192
  def process_invunit_number_data(file_path: str) -> pd.DataFrame:
193
  """
194
  Process and append INVUNIT_NUMBER dataframe to All DB buffers.
@@ -199,7 +211,7 @@ def process_invunit_number_data(file_path: str) -> pd.DataFrame:
199
  return df_invunit_number
200
 
201
 
202
- def process_invunit_data(file_path: str) -> pd.DataFrame:
203
  """
204
  Process data from the specified file path.
205
 
@@ -251,6 +263,18 @@ def process_invunit_data(file_path: str) -> pd.DataFrame:
251
  ).tolist()
252
  ]
253
 
 
 
 
 
 
 
 
 
 
 
 
 
254
  UtilsVars.all_db_dfs.append(df_invunit)
255
  UtilsVars.all_db_dfs_names.append("INVUNIT")
256
  return df_invunit
 
3
  from utils.convert_to_excel import convert_invunit_dfs, save_dataframe
4
  from utils.dump_excel import read_dump_excel
5
  from utils.extract_code import extract_code_from_mrbts
6
+ from utils.processing_cache import get_or_build_processed
7
  from utils.utils_vars import UtilsVars
8
 
9
  RF_UNIT = [
 
113
  return df
114
 
115
 
116
+ def _build_invunit_number_dataframe(file_path: str) -> pd.DataFrame:
117
  """
118
  Build detailed INVUNIT_NUMBER dataframe from dump INVUNIT sheet.
119
  """
 
190
  return df_invunit_number
191
 
192
 
193
+ def build_invunit_number_dataframe(file_path: str) -> pd.DataFrame:
194
+ """
195
+ Build detailed INVUNIT_NUMBER dataframe from dump INVUNIT sheet.
196
+ """
197
+ return get_or_build_processed(
198
+ file_path,
199
+ "invunit_number",
200
+ lambda: _build_invunit_number_dataframe(file_path),
201
+ )
202
+
203
+
204
  def process_invunit_number_data(file_path: str) -> pd.DataFrame:
205
  """
206
  Process and append INVUNIT_NUMBER dataframe to All DB buffers.
 
211
  return df_invunit_number
212
 
213
 
214
+ def _build_invunit_data(file_path: str) -> pd.DataFrame:
215
  """
216
  Process data from the specified file path.
217
 
 
263
  ).tolist()
264
  ]
265
 
266
+ return df_invunit
267
+
268
+
269
+ def process_invunit_data(file_path: str) -> pd.DataFrame:
270
+ """
271
+ Process data from the specified file path.
272
+ """
273
+ df_invunit = get_or_build_processed(
274
+ file_path,
275
+ "invunit",
276
+ lambda: _build_invunit_data(file_path),
277
+ )
278
  UtilsVars.all_db_dfs.append(df_invunit)
279
  UtilsVars.all_db_dfs_names.append("INVUNIT")
280
  return df_invunit
queries/process_lte.py CHANGED
@@ -5,6 +5,7 @@ from utils.config_band import config_band, lte_mrbts_band
5
  from utils.convert_to_excel import convert_dfs, save_dataframe
6
  from utils.dump_excel import read_dump_excel
7
  from utils.kml_creator import generate_kml_from_df
 
8
  from utils.utils_vars import (
9
  LteFddAnalysisData,
10
  LteTddAnalysisData,
@@ -156,7 +157,7 @@ def process_lncel(file_path: str):
156
  return df_lncel
157
 
158
 
159
- def process_lte_data(file_path: str):
160
  """
161
  Process data from the specified file path.
162
 
@@ -237,16 +238,27 @@ def process_lte_data(file_path: str):
237
  # Save dataframes
238
  # save_dataframe(df_fdd_final, "fdd")
239
  # save_dataframe(df_tdd_final, "tdd")
240
- UtilsVars.all_db_dfs.extend([df_fdd_final, df_tdd_final])
241
- UtilsVars.lte_dfs.extend([df_fdd_final, df_tdd_final])
242
- UtilsVars.all_db_dfs_names.extend(["LTE_FDD", "LTE_TDD"])
243
-
244
  return [df_fdd_final, df_tdd_final]
245
  # add the fdd and tdd to the list
246
 
247
  # UtilsVars.final_lte_database = [df_fdd_final, df_tdd_final]
248
 
249
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
250
  def process_lte_data_to_excel(file_path: str):
251
  lte_dfs = process_lte_data(file_path)
252
  UtilsVars.final_lte_database = convert_dfs(lte_dfs, ["LTE_FDD", "LTE_TDD"])
 
5
  from utils.convert_to_excel import convert_dfs, save_dataframe
6
  from utils.dump_excel import read_dump_excel
7
  from utils.kml_creator import generate_kml_from_df
8
+ from utils.processing_cache import get_or_build_processed
9
  from utils.utils_vars import (
10
  LteFddAnalysisData,
11
  LteTddAnalysisData,
 
157
  return df_lncel
158
 
159
 
160
+ def _build_lte_data(file_path: str) -> list[pd.DataFrame]:
161
  """
162
  Process data from the specified file path.
163
 
 
238
  # Save dataframes
239
  # save_dataframe(df_fdd_final, "fdd")
240
  # save_dataframe(df_tdd_final, "tdd")
 
 
 
 
241
  return [df_fdd_final, df_tdd_final]
242
  # add the fdd and tdd to the list
243
 
244
  # UtilsVars.final_lte_database = [df_fdd_final, df_tdd_final]
245
 
246
 
247
+ def process_lte_data(file_path: str):
248
+ """
249
+ Process LTE data and preserve the historical UtilsVars side effects.
250
+ """
251
+ lte_dfs = get_or_build_processed(
252
+ file_path,
253
+ "lte_data",
254
+ lambda: _build_lte_data(file_path),
255
+ )
256
+ UtilsVars.all_db_dfs.extend(lte_dfs)
257
+ UtilsVars.lte_dfs.extend(lte_dfs)
258
+ UtilsVars.all_db_dfs_names.extend(["LTE_FDD", "LTE_TDD"])
259
+ return lte_dfs
260
+
261
+
262
  def process_lte_data_to_excel(file_path: str):
263
  lte_dfs = process_lte_data(file_path)
264
  UtilsVars.final_lte_database = convert_dfs(lte_dfs, ["LTE_FDD", "LTE_TDD"])
queries/process_trx.py CHANGED
@@ -141,24 +141,12 @@ def process_trx_with_bts_name(
141
  # TCHs SDs BCCH CCCH CBC Total Signal
142
 
143
  # Calculate "count of channels per TRX" for each row
144
- df_trx_bts_name["TCHs"] = df_trx_bts_name[channel_columns].apply(
145
- lambda row: (row == 2).sum(), axis=1
146
- )
147
- df_trx_bts_name["SDs"] = df_trx_bts_name[channel_columns].apply(
148
- lambda row: (row == 3).sum(), axis=1
149
- )
150
-
151
- df_trx_bts_name["BCCHs"] = df_trx_bts_name[channel_columns].apply(
152
- lambda row: (row == 4).sum(), axis=1
153
- )
154
-
155
- df_trx_bts_name["CCCHs"] = df_trx_bts_name[channel_columns].apply(
156
- lambda row: (row == 6).sum(), axis=1
157
- )
158
-
159
- df_trx_bts_name["CBCs"] = df_trx_bts_name[channel_columns].apply(
160
- lambda row: (row == 8).sum(), axis=1
161
- )
162
 
163
  # Total Channels = TCHs + SDs + BCCHs + CCCHs + CBCs
164
 
@@ -175,27 +163,20 @@ def process_trx_with_bts_name(
175
  df_trx_bts_name["Signal"] = (
176
  df_trx_bts_name["BCCHs"] + df_trx_bts_name["CCCHs"] + df_trx_bts_name["CBCs"]
177
  )
178
- df_trx_bts_name["number_tch_per_cell"] = df_trx_bts_name.groupby("ID_BTS")[
179
- "TCHs"
180
- ].transform("sum")
181
- df_trx_bts_name["number_sd_per_cell"] = df_trx_bts_name.groupby("ID_BTS")[
182
- "SDs"
183
- ].transform("sum")
184
- df_trx_bts_name["number_bcch_per_cell"] = df_trx_bts_name.groupby("ID_BTS")[
185
- "BCCHs"
186
- ].transform("sum")
187
- df_trx_bts_name["number_ccch_per_cell"] = df_trx_bts_name.groupby("ID_BTS")[
188
- "CCCHs"
189
- ].transform("sum")
190
- df_trx_bts_name["number_cbc_per_cell"] = df_trx_bts_name.groupby("ID_BTS")[
191
- "CBCs"
192
- ].transform("sum")
193
- df_trx_bts_name["number_total_channels_per_cell"] = df_trx_bts_name.groupby(
194
- "ID_BTS"
195
- )["TotalChannels"].transform("sum")
196
- df_trx_bts_name["number_signals_per_cell"] = df_trx_bts_name.groupby("ID_BTS")[
197
- "Signal"
198
  ].transform("sum")
 
 
199
 
200
  # Avoir les TRX par bande et par secteur et BCF sous forme concaténée comme 5/3/3
201
 
 
141
  # TCHs SDs BCCH CCCH CBC Total Signal
142
 
143
  # Calculate "count of channels per TRX" for each row
144
+ channel_data = df_trx_bts_name[channel_columns]
145
+ df_trx_bts_name["TCHs"] = channel_data.eq(2).sum(axis=1)
146
+ df_trx_bts_name["SDs"] = channel_data.eq(3).sum(axis=1)
147
+ df_trx_bts_name["BCCHs"] = channel_data.eq(4).sum(axis=1)
148
+ df_trx_bts_name["CCCHs"] = channel_data.eq(6).sum(axis=1)
149
+ df_trx_bts_name["CBCs"] = channel_data.eq(8).sum(axis=1)
 
 
 
 
 
 
 
 
 
 
 
 
150
 
151
  # Total Channels = TCHs + SDs + BCCHs + CCCHs + CBCs
152
 
 
163
  df_trx_bts_name["Signal"] = (
164
  df_trx_bts_name["BCCHs"] + df_trx_bts_name["CCCHs"] + df_trx_bts_name["CBCs"]
165
  )
166
+ channel_total_columns = {
167
+ "TCHs": "number_tch_per_cell",
168
+ "SDs": "number_sd_per_cell",
169
+ "BCCHs": "number_bcch_per_cell",
170
+ "CCCHs": "number_ccch_per_cell",
171
+ "CBCs": "number_cbc_per_cell",
172
+ "TotalChannels": "number_total_channels_per_cell",
173
+ "Signal": "number_signals_per_cell",
174
+ }
175
+ totals_per_cell = df_trx_bts_name.groupby("ID_BTS")[
176
+ list(channel_total_columns)
 
 
 
 
 
 
 
 
 
177
  ].transform("sum")
178
+ totals_per_cell.rename(columns=channel_total_columns, inplace=True)
179
+ df_trx_bts_name = pd.concat([df_trx_bts_name, totals_per_cell], axis=1)
180
 
181
  # Avoir les TRX par bande et par secteur et BCF sous forme concaténée comme 5/3/3
182
 
queries/process_wcdma.py CHANGED
@@ -5,6 +5,7 @@ from utils.convert_to_excel import convert_dfs, save_dataframe
5
  from utils.dump_excel import read_dump_excel
6
  from utils.extract_code import extract_code_from_mrbts
7
  from utils.kml_creator import generate_kml_from_df
 
8
  from utils.utils_vars import UtilsVars, WcdmaAnalysisData, get_physical_db
9
 
10
  WCEL_COLUMNS = [
@@ -94,7 +95,7 @@ WCDMA_KML_COLUMNS = [
94
  ]
95
 
96
 
97
- def process_wcdma_data(file_path: str):
98
  """
99
  Process data from the specified file path.
100
 
@@ -190,17 +191,27 @@ def process_wcdma_data(file_path: str):
190
  # save_dataframe(df_wcel_bcf, "wbts")
191
  # save_dataframe(df_wncel, "wncel")
192
  # df_3g = save_dataframe(df_3g, "3G")
193
- UtilsVars.all_db_dfs.append(df_3g)
194
- UtilsVars.wcdma_dfs.append(df_3g)
195
- UtilsVars.all_db_dfs_names.append("WCDMA")
196
-
197
- # UtilsVars.final_wcdma_database = convert_dfs([df_3g], ["WCDMA"])
198
  return df_3g
199
  # UtilsVars.final_wcdma_database = [df_3g]
200
 
201
  # BTS.process_ok = "Done"
202
 
203
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
204
  def process_wcdma_data_to_excel(file_path: str):
205
  """
206
  Process WCDMA data from the specified file path and convert it to Excel format
 
5
  from utils.dump_excel import read_dump_excel
6
  from utils.extract_code import extract_code_from_mrbts
7
  from utils.kml_creator import generate_kml_from_df
8
+ from utils.processing_cache import get_or_build_processed
9
  from utils.utils_vars import UtilsVars, WcdmaAnalysisData, get_physical_db
10
 
11
  WCEL_COLUMNS = [
 
95
  ]
96
 
97
 
98
+ def _build_wcdma_data(file_path: str) -> pd.DataFrame:
99
  """
100
  Process data from the specified file path.
101
 
 
191
  # save_dataframe(df_wcel_bcf, "wbts")
192
  # save_dataframe(df_wncel, "wncel")
193
  # df_3g = save_dataframe(df_3g, "3G")
 
 
 
 
 
194
  return df_3g
195
  # UtilsVars.final_wcdma_database = [df_3g]
196
 
197
  # BTS.process_ok = "Done"
198
 
199
 
200
+ def process_wcdma_data(file_path: str):
201
+ """
202
+ Process WCDMA data and preserve the historical UtilsVars side effects.
203
+ """
204
+ df_3g = get_or_build_processed(
205
+ file_path,
206
+ "wcdma_data",
207
+ lambda: _build_wcdma_data(file_path),
208
+ )
209
+ UtilsVars.all_db_dfs.append(df_3g)
210
+ UtilsVars.wcdma_dfs.append(df_3g)
211
+ UtilsVars.all_db_dfs_names.append("WCDMA")
212
+ return df_3g
213
+
214
+
215
  def process_wcdma_data_to_excel(file_path: str):
216
  """
217
  Process WCDMA data from the specified file path and convert it to Excel format
scripts/benchmark_database_processing.py ADDED
@@ -0,0 +1,110 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import gc
5
+ import sys
6
+ import time
7
+ import tracemalloc
8
+ from collections.abc import Callable
9
+ from pathlib import Path
10
+
11
+ ROOT = Path(__file__).resolve().parents[1]
12
+ if str(ROOT) not in sys.path:
13
+ sys.path.insert(0, str(ROOT))
14
+
15
+ from queries.process_all_db import clear_all_dbs, process_all_tech_db, process_atoll_db, process_nice_db
16
+ from queries.process_gsm import process_gsm_data_to_excel
17
+ from queries.process_lte import process_lte_data_to_excel
18
+ from utils.dump_excel import clear_dump_excel_cache
19
+ from utils.processing_cache import clear_processed_dataframe_cache
20
+ from utils.utils_vars import UtilsVars
21
+
22
+
23
+ def reset_outputs(*, clear_caches: bool) -> None:
24
+ clear_all_dbs()
25
+ UtilsVars.final_gsm_database = ""
26
+ UtilsVars.final_lte_database = ""
27
+ UtilsVars.final_nice_database = None
28
+ UtilsVars.final_atoll_database = None
29
+ if clear_caches:
30
+ clear_dump_excel_cache()
31
+ clear_processed_dataframe_cache()
32
+ gc.collect()
33
+
34
+
35
+ def run_case(
36
+ label: str,
37
+ func: Callable[[str], None],
38
+ dump_path: str,
39
+ *,
40
+ clear_caches: bool,
41
+ measure_memory: bool,
42
+ ) -> dict[str, float | str]:
43
+ reset_outputs(clear_caches=clear_caches)
44
+ if measure_memory:
45
+ tracemalloc.start()
46
+ start = time.perf_counter()
47
+ func(dump_path)
48
+ elapsed = time.perf_counter() - start
49
+ peak_mib = 0.0
50
+ if measure_memory:
51
+ _, peak = tracemalloc.get_traced_memory()
52
+ tracemalloc.stop()
53
+ peak_mib = peak / 1024 / 1024
54
+ return {"case": label, "seconds": elapsed, "peak_mib": peak_mib}
55
+
56
+
57
+ def print_result(result: dict[str, float | str], *, measure_memory: bool) -> None:
58
+ memory = f" peak={result['peak_mib']:.1f} MiB" if measure_memory else ""
59
+ print(f"{result['case']}: {result['seconds']:.2f}s{memory}", flush=True)
60
+
61
+
62
+ def main() -> None:
63
+ parser = argparse.ArgumentParser(description="Benchmark OML_DB database generation paths.")
64
+ parser.add_argument("dump", type=Path, help="Path to a dump .xlsb file.")
65
+ parser.add_argument(
66
+ "--memory",
67
+ action="store_true",
68
+ help="Measure approximate Python allocation peak with tracemalloc. Slower.",
69
+ )
70
+ args = parser.parse_args()
71
+ dump_path = str(args.dump.expanduser().resolve())
72
+
73
+ cases: list[tuple[str, Callable[[str], None]]] = [
74
+ ("Generate 2G DB", process_gsm_data_to_excel),
75
+ ("Generate LTE DB", process_lte_data_to_excel),
76
+ ("Generate All DBs", process_all_tech_db),
77
+ ("Generate Nice DB", process_nice_db),
78
+ ("Generate Atoll DB", process_atoll_db),
79
+ ]
80
+
81
+ for label, func in cases:
82
+ print_result(
83
+ run_case(label, func, dump_path, clear_caches=True, measure_memory=args.memory),
84
+ measure_memory=args.memory,
85
+ )
86
+
87
+ print_result(
88
+ run_case(
89
+ "Generate All DBs repeat #1",
90
+ process_all_tech_db,
91
+ dump_path,
92
+ clear_caches=True,
93
+ measure_memory=args.memory,
94
+ ),
95
+ measure_memory=args.memory,
96
+ )
97
+ print_result(
98
+ run_case(
99
+ "Generate All DBs repeat #2 same process",
100
+ process_all_tech_db,
101
+ dump_path,
102
+ clear_caches=False,
103
+ measure_memory=args.memory,
104
+ ),
105
+ measure_memory=args.memory,
106
+ )
107
+
108
+
109
+ if __name__ == "__main__":
110
+ main()
tests/test_processing_reuse.py ADDED
@@ -0,0 +1,180 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import pandas as pd
2
+
3
+ from utils.processing_cache import clear_processed_dataframe_cache
4
+ from utils.utils_vars import UtilsVars
5
+
6
+
7
+ def setup_function():
8
+ clear_processed_dataframe_cache()
9
+ UtilsVars.all_db_dfs = []
10
+ UtilsVars.all_db_dfs_names = []
11
+ UtilsVars.gsm_dfs = []
12
+ UtilsVars.wcdma_dfs = []
13
+ UtilsVars.lte_dfs = []
14
+
15
+
16
+ def assert_same_frame(left: pd.DataFrame, right: pd.DataFrame) -> None:
17
+ pd.testing.assert_frame_equal(
18
+ left.reset_index(drop=True),
19
+ right.reset_index(drop=True),
20
+ check_dtype=True,
21
+ check_like=False,
22
+ )
23
+
24
+
25
+ def test_lte_processed_cache_preserves_frames_and_side_effects(monkeypatch):
26
+ from queries import process_lte
27
+
28
+ calls = {"count": 0}
29
+ expected_fdd = pd.DataFrame({"A": pd.Series([1, None], dtype="Int64"), "B": ["x", None]})
30
+ expected_tdd = pd.DataFrame({"C": pd.Series([1.5, None], dtype="Float64"), "D": ["y", None]})
31
+
32
+ def fake_build(file_path):
33
+ calls["count"] += 1
34
+ return [expected_fdd.copy(deep=True), expected_tdd.copy(deep=True)]
35
+
36
+ monkeypatch.setattr(process_lte, "_build_lte_data", fake_build)
37
+
38
+ first_fdd, first_tdd = process_lte.process_lte_data("dump.xlsb")
39
+ first_fdd.loc[0, "B"] = "mutated"
40
+
41
+ second_fdd, second_tdd = process_lte.process_lte_data("dump.xlsb")
42
+
43
+ assert calls["count"] == 1
44
+ assert_same_frame(second_fdd, expected_fdd)
45
+ assert_same_frame(first_tdd, expected_tdd)
46
+ assert_same_frame(second_tdd, expected_tdd)
47
+ assert UtilsVars.all_db_dfs_names == ["LTE_FDD", "LTE_TDD", "LTE_FDD", "LTE_TDD"]
48
+
49
+
50
+ def test_wcdma_processed_cache_preserves_frame_and_side_effects(monkeypatch):
51
+ from queries import process_wcdma
52
+
53
+ calls = {"count": 0}
54
+ expected = pd.DataFrame(
55
+ {
56
+ "ID_WCEL": pd.Series(["1_2_3", None], dtype="string"),
57
+ "LAC": pd.Series([10, None], dtype="Int64"),
58
+ }
59
+ )
60
+
61
+ def fake_build(file_path):
62
+ calls["count"] += 1
63
+ return expected.copy(deep=True)
64
+
65
+ monkeypatch.setattr(process_wcdma, "_build_wcdma_data", fake_build)
66
+
67
+ first = process_wcdma.process_wcdma_data("dump.xlsb")
68
+ first.loc[0, "ID_WCEL"] = "mutated"
69
+ second = process_wcdma.process_wcdma_data("dump.xlsb")
70
+
71
+ assert calls["count"] == 1
72
+ assert_same_frame(second, expected)
73
+ assert UtilsVars.all_db_dfs_names == ["WCDMA", "WCDMA"]
74
+
75
+
76
+ def test_gsm_bundle_cache_preserves_gsm_mal_trx(monkeypatch):
77
+ from queries import process_gsm
78
+
79
+ calls = {"count": 0}
80
+ gsm = pd.DataFrame({"ID_BTS": pd.Series(["1_2_3"], dtype="string"), "BCCH": pd.Series([10], dtype="Int64")})
81
+ mal = pd.DataFrame({"ID_MAL": pd.Series(["1_4"], dtype="string"), "MAL_TCH": ["1,2"]})
82
+ trx = pd.DataFrame({"TRX": pd.Series([1], dtype="Int64"), "name": ["BTS_A"]})
83
+
84
+ def fake_build(file_path):
85
+ calls["count"] += 1
86
+ return [gsm.copy(deep=True), mal.copy(deep=True), trx.copy(deep=True)]
87
+
88
+ monkeypatch.setattr(process_gsm, "_build_combined_gsm_database", fake_build)
89
+
90
+ first = process_gsm.combined_gsm_database("dump.xlsb")
91
+ first[0].loc[0, "ID_BTS"] = "mutated"
92
+ second = process_gsm.combined_gsm_database("dump.xlsb")
93
+
94
+ assert calls["count"] == 1
95
+ for actual, expected in zip(second, [gsm, mal, trx]):
96
+ assert_same_frame(actual, expected)
97
+ assert UtilsVars.all_db_dfs_names == ["GSM", "MAL", "TRX", "GSM", "MAL", "TRX"]
98
+
99
+
100
+ def test_invunit_caches_preserve_nulls_columns_and_dtypes(monkeypatch):
101
+ from queries import process_invunit
102
+
103
+ invunit_calls = {"count": 0}
104
+ number_calls = {"count": 0}
105
+ invunit = pd.DataFrame(
106
+ {
107
+ "MRBTS": pd.Series(["12345", None], dtype="string"),
108
+ "code": pd.Series([123, None], dtype="Int64"),
109
+ "invunit_summary": pd.Series(["1 FBBA", None], dtype="string"),
110
+ }
111
+ )
112
+ invunit_number = pd.DataFrame(
113
+ {
114
+ "MRBTS": pd.Series(["12345", None], dtype="string"),
115
+ "name": pd.Series(["SITE_A", None], dtype="string"),
116
+ "inventoryUnitType": pd.Series(["FBBA", None], dtype="string"),
117
+ "vendorUnitTypeNumber": pd.Series(["V1", None], dtype="string"),
118
+ "serialNumber": pd.Series(["S1", None], dtype="string"),
119
+ }
120
+ )
121
+
122
+ def fake_invunit(file_path):
123
+ invunit_calls["count"] += 1
124
+ return invunit.copy(deep=True)
125
+
126
+ def fake_invunit_number(file_path):
127
+ number_calls["count"] += 1
128
+ return invunit_number.copy(deep=True)
129
+
130
+ monkeypatch.setattr(process_invunit, "_build_invunit_data", fake_invunit)
131
+ monkeypatch.setattr(process_invunit, "_build_invunit_number_dataframe", fake_invunit_number)
132
+
133
+ first = process_invunit.process_invunit_data("dump.xlsb")
134
+ first.loc[0, "invunit_summary"] = "mutated"
135
+ second = process_invunit.process_invunit_data("dump.xlsb")
136
+ number_first = process_invunit.build_invunit_number_dataframe("dump.xlsb")
137
+ number_first.loc[0, "serialNumber"] = "mutated"
138
+ number_second = process_invunit.build_invunit_number_dataframe("dump.xlsb")
139
+
140
+ assert invunit_calls["count"] == 1
141
+ assert number_calls["count"] == 1
142
+ assert_same_frame(second, invunit)
143
+ assert_same_frame(number_second, invunit_number)
144
+
145
+
146
+ def test_physical_db_cache_invalidates_when_file_changes(tmp_path, monkeypatch):
147
+ import utils.utils_vars as utils_vars
148
+
149
+ physical_path = tmp_path / "physical_database.csv"
150
+ columns = [
151
+ "Code_Sector",
152
+ "Azimut",
153
+ "Longitude",
154
+ "Latitude",
155
+ "Hauteur",
156
+ "City",
157
+ "Adresse",
158
+ "Commune",
159
+ "Cercle",
160
+ ]
161
+ pd.DataFrame([["1_1", 10, -1.0, 2.0, 30, "A", "Addr", "Com", "Cer"]], columns=columns).to_csv(
162
+ physical_path,
163
+ index=False,
164
+ )
165
+ monkeypatch.setattr(utils_vars, "url", str(physical_path))
166
+ monkeypatch.setattr(utils_vars, "_PHYSICAL_DB_CACHE", None)
167
+
168
+ first = utils_vars.get_physical_db()
169
+ first.loc[0, "City"] = "mutated"
170
+ second = utils_vars.get_physical_db()
171
+ assert second.loc[0, "City"] == "A"
172
+
173
+ pd.DataFrame([["1_1", 20, -1.0, 2.0, 30, "B_LONG", "Addr", "Com", "Cer"]], columns=columns).to_csv(
174
+ physical_path,
175
+ index=False,
176
+ )
177
+
178
+ third = utils_vars.get_physical_db()
179
+ assert third.loc[0, "Azimut"] == 20
180
+ assert third.loc[0, "City"] == "B_LONG"
utils/dump_excel.py CHANGED
@@ -67,6 +67,13 @@ def _file_cache_key(file_path) -> tuple:
67
  return ("object", id(file_path), name, size)
68
 
69
 
 
 
 
 
 
 
 
70
  def _freeze_kwargs(kwargs: dict) -> tuple:
71
  return tuple(sorted((key, repr(value)) for key, value in kwargs.items()))
72
 
 
67
  return ("object", id(file_path), name, size)
68
 
69
 
70
+ def get_dump_file_cache_key(file_path) -> tuple:
71
+ """
72
+ Return the cache identity used for a dump path/upload object.
73
+ """
74
+ return _file_cache_key(file_path)
75
+
76
+
77
  def _freeze_kwargs(kwargs: dict) -> tuple:
78
  return tuple(sorted((key, repr(value)) for key, value in kwargs.items()))
79
 
utils/processing_cache.py ADDED
@@ -0,0 +1,76 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from __future__ import annotations
2
+
3
+ from collections import OrderedDict
4
+ from collections.abc import Callable
5
+ from typing import TypeVar
6
+
7
+ import pandas as pd
8
+
9
+ from utils.dump_excel import get_dump_file_cache_key
10
+ from utils.utils_vars import UtilsVars
11
+
12
+
13
+ T = TypeVar("T")
14
+
15
+ _MAX_PROCESSED_CACHE_ITEMS = 16
16
+ _PROCESSED_CACHE: OrderedDict[tuple, object] = OrderedDict()
17
+
18
+
19
+ def _clone(value):
20
+ if isinstance(value, pd.DataFrame):
21
+ return value.copy(deep=True)
22
+ if isinstance(value, list):
23
+ return [_clone(item) for item in value]
24
+ if isinstance(value, tuple):
25
+ return tuple(_clone(item) for item in value)
26
+ if isinstance(value, dict):
27
+ return {key: _clone(item) for key, item in value.items()}
28
+ return value
29
+
30
+
31
+ def _remember(key: tuple, value) -> None:
32
+ _PROCESSED_CACHE[key] = _clone(value)
33
+ _PROCESSED_CACHE.move_to_end(key)
34
+ while len(_PROCESSED_CACHE) > _MAX_PROCESSED_CACHE_ITEMS:
35
+ _PROCESSED_CACHE.popitem(last=False)
36
+
37
+
38
+ def _cache_key(file_path, namespace: str, extra: tuple = ()) -> tuple:
39
+ return (
40
+ namespace,
41
+ get_dump_file_cache_key(file_path),
42
+ bool(UtilsVars.exclude_decommissioned_2g_bsc),
43
+ tuple(sorted(UtilsVars.decommissioned_2g_bsc_ids)),
44
+ extra,
45
+ )
46
+
47
+
48
+ def clear_processed_dataframe_cache() -> None:
49
+ _PROCESSED_CACHE.clear()
50
+
51
+
52
+ def get_or_build_processed(
53
+ file_path,
54
+ namespace: str,
55
+ builder: Callable[[], T],
56
+ *,
57
+ extra: tuple = (),
58
+ ) -> T:
59
+ key = _cache_key(file_path, namespace, extra)
60
+ if key in _PROCESSED_CACHE:
61
+ _PROCESSED_CACHE.move_to_end(key)
62
+ return _clone(_PROCESSED_CACHE[key])
63
+
64
+ value = builder()
65
+ _remember(key, value)
66
+ return _clone(value)
67
+
68
+
69
+ def remember_processed(
70
+ file_path,
71
+ namespace: str,
72
+ value: T,
73
+ *,
74
+ extra: tuple = (),
75
+ ) -> None:
76
+ _remember(_cache_key(file_path, namespace, extra), value)
utils/utils_vars.py CHANGED
@@ -1,8 +1,10 @@
1
  import numpy as np
2
  import pandas as pd
 
3
 
4
  # url = "https://raw.githubusercontent.com/DavMelchi/STORAGE/refs/heads/main/physical_db/physical_database.csv"
5
  url = r"./physical_db/physical_database.csv"
 
6
 
7
 
8
  def get_physical_db():
@@ -14,7 +16,14 @@ def get_physical_db():
14
  Returns:
15
  pd.DataFrame: A DataFrame containing the filtered columns.
16
  """
17
- physical = pd.read_csv(url)
 
 
 
 
 
 
 
18
  physical = physical[
19
  [
20
  "Code_Sector",
@@ -28,7 +37,8 @@ def get_physical_db():
28
  "Cercle",
29
  ]
30
  ]
31
- return physical
 
32
 
33
 
34
  class UtilsVars:
 
1
  import numpy as np
2
  import pandas as pd
3
+ from pathlib import Path
4
 
5
  # url = "https://raw.githubusercontent.com/DavMelchi/STORAGE/refs/heads/main/physical_db/physical_database.csv"
6
  url = r"./physical_db/physical_database.csv"
7
+ _PHYSICAL_DB_CACHE: tuple[tuple[str, int, int], pd.DataFrame] | None = None
8
 
9
 
10
  def get_physical_db():
 
16
  Returns:
17
  pd.DataFrame: A DataFrame containing the filtered columns.
18
  """
19
+ global _PHYSICAL_DB_CACHE
20
+ path = Path(url)
21
+ stat = path.stat()
22
+ cache_key = (str(path.resolve()), stat.st_size, stat.st_mtime_ns)
23
+ if _PHYSICAL_DB_CACHE is not None and _PHYSICAL_DB_CACHE[0] == cache_key:
24
+ return _PHYSICAL_DB_CACHE[1].copy(deep=True)
25
+
26
+ physical = pd.read_csv(path)
27
  physical = physical[
28
  [
29
  "Code_Sector",
 
37
  "Cercle",
38
  ]
39
  ]
40
+ _PHYSICAL_DB_CACHE = (cache_key, physical)
41
+ return physical.copy(deep=True)
42
 
43
 
44
  class UtilsVars: