Skip to content
Merged
4 changes: 2 additions & 2 deletions asap-tools/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -115,7 +115,7 @@ experiments:
With this config, 2 experiments are run independently. In the first experiment, `asap-tools/queriers/prometheus-client` only sends queries to ASAP. After this experiment finishes, the infra is torn down. Then the second experiment is set up and `asap-tools/queriers/prometheus-client` sends queries only to Prometheus directly. In the second experiment (i.e. when `mode=prometheus`), none of ASAP's components are set up (apart from `asap-tools/queriers/prometheus-client`).

Post-experiment analysis:
- Use `compare_costs.py` and `compare_latencies.py` from `$REPO_DIR/asap-tools/experiments/post_experiments/`.
- Use `compare_costs.py` and `compare_latencies.py` from `$REPO_DIR/asap-tools/experiments/post_experiment/single_experiment/`.
- `run_compare_latencies.sh` is an easy wrapper around `compare_latencies.py`

### Comparing query accuracy for ASAP vs Prometheus
Expand All @@ -130,7 +130,7 @@ experiments:
With this config, only one experiment is run. In the same experiment, `PrometheusClient` sends a query to ASAP and then immediately after that, sends a query to Prometheus too.

Post-experiment analysis:
- Use `calculate_fidelity.py` from `$REPO_DIR/asap-tools/experiments/post_experiments/`.
- Use `calculate_fidelity.py` from `$REPO_DIR/asap-tools/experiments/post_experiment/single_experiment/`.
- `run_calculate_fidelity.sh` is an easy wrapper around `calculate_fidelity.py`

### Debugging with Verbose Logging
Expand Down
8 changes: 8 additions & 0 deletions asap-tools/experiments/post_experiment/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
# post_experiment

- `single_experiment/` — analyze one experiment (its `baseline` and `sketchdb` modes). Prints stats, `--machine-readable` JSON, optional plots. Cost and latency definitions live here.
- `multi_experiment/` — sweep across experiments and produce comparison figures. Get numbers from `single_experiment/` scripts or `lib/`.
- `lib/` — shared loaders (`results_loader.py`).
- `debug/` — one-off inspection tools.

Run scripts from any directory; they locate `constants.py` and sibling scripts relative to their own path.
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@
import os
import sys
import re
import json
import glob
import yaml
import argparse
Expand All @@ -45,13 +46,17 @@
)

# Add parent directories to path for imports
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
sys.path.append(
os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
)
import constants # noqa: E402
from post_experiment.results_loader import ( # noqa: E402
from post_experiment.lib.results_loader import ( # noqa: E402
load_latencies_only,
get_server_name_for_mode,
)
from post_experiment.compare_latencies import calculate_latency_stats # noqa: E402
from post_experiment.single_experiment.compare_latencies import ( # noqa: E402
calculate_latency_stats,
)

# Metric mapping for cost benefit (compare_costs.py doesn't have 'mean')
METRIC_TO_CPU_STAT = {
Expand Down Expand Up @@ -147,16 +152,15 @@ def load_experiment_config(exp_dir: str) -> Dict[str, Any]:
return yaml.safe_load(f)


def extract_experiment_data(
exp_name: str, metric: str = "p95", verify_scale: bool = True
) -> Optional[Dict[str, Any]]:
def extract_experiment_data(exp_name: str, metric: str) -> Optional[Dict[str, Any]]:
"""
Extract data from a single experiment.

Warns if the data scale doesn't match the expected 2^card_exp.

Args:
exp_name: Experiment name
metric: Latency metric to use (median, p95, p99, mean)
verify_scale: If True, verify data scale matches expected 2^card_exp

Returns:
dict with experiment data or None if extraction fails
Expand All @@ -181,7 +185,7 @@ def extract_experiment_data(
actual_scale = calculate_data_scale_from_config(config)
expected_scale = 2 ** metadata["card_exp"]

if verify_scale and actual_scale != expected_scale:
if actual_scale != expected_scale:
print(
f"Warning: {exp_name} has scale {actual_scale} but expected {expected_scale}"
)
Expand Down Expand Up @@ -257,17 +261,18 @@ def extract_experiment_data(


def extract_cost_benefit_data(
exp_name: str, metric: str = "p95", verify_scale: bool = True
exp_name: str, metric: str, total: bool
) -> Optional[Dict[str, Any]]:
"""
Extract cost benefit data from a single experiment.

Runs compare_costs.py and parses Query CPU Benefit output.
Runs compare_costs.py and reads the baseline/sketchdb CPU ratio. Warns if
the data scale doesn't match the expected 2^card_exp.

Args:
exp_name: Experiment name
metric: CPU metric to use (median, p95, p99, sum, max)
verify_scale: If True, verify data scale matches expected 2^card_exp
total: Use total CPU (all processes) instead of query CPU

Returns:
dict with experiment data or None if extraction fails
Expand All @@ -285,19 +290,19 @@ def extract_cost_benefit_data(
return None

try:
# Verify scale if requested
actual_scale = None
if verify_scale:
config = load_experiment_config(exp_dir)
actual_scale = calculate_data_scale_from_config(config)
expected_scale = 2 ** metadata["card_exp"]
if actual_scale != expected_scale:
print(
f"Warning: {exp_name} has scale {actual_scale} but expected {expected_scale}"
)
config = load_experiment_config(exp_dir)
actual_scale = calculate_data_scale_from_config(config)
expected_scale = 2 ** metadata["card_exp"]
if actual_scale != expected_scale:
print(
f"Warning: {exp_name} has scale {actual_scale} but expected {expected_scale}"
)

# Run compare_costs.py
script_dir = os.path.dirname(os.path.abspath(__file__))
script_dir = os.path.join(
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
"single_experiment",
)
compare_costs_path = os.path.join(script_dir, "compare_costs.py")

result = subprocess.run(
Expand All @@ -308,29 +313,24 @@ def extract_cost_benefit_data(
exp_name,
"--all_experiment_modes",
"--print",
"--machine-readable",
],
capture_output=True,
text=True,
check=True,
cwd=script_dir,
)

# Parse output for Query CPU Benefit section
output = result.stdout + result.stderr
costs = json.loads(result.stdout)
cpu_stat = METRIC_TO_CPU_STAT.get(metric, "p95")

# Look for pattern: " <stat>: <value>x" in Query CPU Benefit section
pattern = rf"Query CPU Benefit.*?^\s+{cpu_stat}:\s+([\d.]+)x"
match = re.search(pattern, output, re.MULTILINE | re.DOTALL)

if not match:
print(
f"Warning: Could not find Query CPU Benefit '{cpu_stat}' for {exp_name}"
)
benefits = (
costs["benefit"]["cpu_percent"] if total else costs["query_cpu_benefit"]
)
benefit_ratio = benefits[cpu_stat]
if np.isnan(benefit_ratio): # 0/0: both modes had zero cost
print(f"Warning: {cpu_stat} CPU benefit is NaN for {exp_name}")
return None

benefit_ratio = float(match.group(1))

return {
"experiment_name": exp_name,
"query_type": metadata["query_type"],
Expand All @@ -352,9 +352,9 @@ def extract_cost_benefit_data(

def extract_experiments_from_patterns(
patterns: List[str],
metric: str = "p95",
cardinalities: Optional[List[int]] = None,
benefit_type: str = "latency",
metric: str,
cardinalities: Optional[List[int]],
benefit_type: str,
) -> pd.DataFrame:
"""
Extract data from experiments matching glob patterns.
Expand All @@ -363,7 +363,7 @@ def extract_experiments_from_patterns(
patterns: List of glob patterns for experiment names
metric: Metric to use (latency or CPU stat depending on benefit_type)
cardinalities: Optional list of cardinality exponents to include
benefit_type: Type of benefit to extract ('latency' or 'cost')
benefit_type: 'latency', 'cost' (query CPU) or 'total_cost' (total CPU)

Returns:
DataFrame with experiment data
Expand All @@ -386,8 +386,10 @@ def extract_experiments_from_patterns(
for exp_name in sorted(exp_names):
if benefit_type == "latency":
exp_data = extract_experiment_data(exp_name, metric=metric)
elif benefit_type == "cost":
exp_data = extract_cost_benefit_data(exp_name, metric=metric)
elif benefit_type in ("cost", "total_cost"):
exp_data = extract_cost_benefit_data(
exp_name, metric=metric, total=benefit_type == "total_cost"
)
else:
raise ValueError(f"Unknown benefit_type: {benefit_type}")

Expand All @@ -410,9 +412,7 @@ def extract_experiments_from_patterns(
return df


def create_plot(
df: pd.DataFrame, metric: str = "p95", benefit_type: str = "latency"
) -> "ggplot":
def create_plot(df: pd.DataFrame, metric: str, benefit_type: str) -> "ggplot":
"""
Create benefit vs lookback plot with log2(T/15) x-axis.

Expand Down Expand Up @@ -456,7 +456,11 @@ def create_plot(
y_breaks = sorted(list(set(y_breaks))) # Remove duplicates and sort

# Dynamic Y-axis label based on benefit type
y_label = "Latency Benefit" if benefit_type == "latency" else "Cost Benefit (CPU)"
y_label = {
"latency": "Latency Benefit",
"cost": "Query CPU Benefit",
"total_cost": "Total CPU Benefit",
}[benefit_type]

p = (
ggplot(
Expand Down Expand Up @@ -494,16 +498,15 @@ def create_plot(
return p


def print_summary_table(
df: pd.DataFrame, metric: str = "p95", benefit_type: str = "latency"
):
def print_summary_table(df: pd.DataFrame, metric: str, benefit_type: str):
"""Print summary table of experiment data."""
# Dynamic header based on benefit type
if benefit_type == "latency":
header = f"Latency Benefit Analysis Summary ({metric.upper()} metric)"
else:
cpu_stat = METRIC_TO_CPU_STAT.get(metric, metric)
header = f"Cost Benefit Analysis Summary (CPU {cpu_stat.upper()})"
cpu_kind = "Total" if benefit_type == "total_cost" else "Query"
header = f"{cpu_kind} CPU Benefit Analysis Summary ({cpu_stat.upper()})"

print("\n" + "=" * 100)
print(header)
Expand Down Expand Up @@ -591,8 +594,11 @@ def main():
"--benefit-type",
type=str,
default="latency",
choices=["latency", "cost"],
help="Type of benefit to plot: latency or cost (CPU) (default: latency)",
choices=["latency", "cost", "total_cost"],
help=(
"Type of benefit to plot: latency, cost (query CPU) or total_cost "
"(total CPU across all processes) (default: latency)"
),
)
parser.add_argument(
"--cardinalities",
Expand Down Expand Up @@ -620,7 +626,7 @@ def main():
parser.error("Must specify at least one of --print or --plot")

# Validate metric compatibility with benefit type
if args.benefit_type == "cost" and args.metric == "mean":
if args.benefit_type != "latency" and args.metric == "mean":
parser.error(
"'mean' metric is not available for cost benefit. Use median, p95, p99, sum, or max"
)
Expand Down
Loading
Loading