import sys
from pathlib import Path
PROJECT_ROOT = Path.cwd().resolve()
while PROJECT_ROOT != PROJECT_ROOT.parent and not (PROJECT_ROOT / "pyproject.toml").exists():
PROJECT_ROOT = PROJECT_ROOT.parent
if not (PROJECT_ROOT / "pyproject.toml").exists():
raise RuntimeError(
"Notebook must be executed within a FHOPS checkout (pyproject.toml not found)."
)
if str(PROJECT_ROOT) not in sys.path:
sys.path.insert(0, str(PROJECT_ROOT))
import pandas as pd
from IPython.display import display
from docs.examples.analytics import utils
from fhops.scenario.synthetic import SyntheticDatasetConfig
SCENARIOS = [
(
"small",
PROJECT_ROOT / "examples/synthetic/small/scenario.yaml",
PROJECT_ROOT / "docs/examples/analytics/data/synthetic_small_sa_assignments.csv",
),
(
"medium",
PROJECT_ROOT / "examples/synthetic/medium/scenario.yaml",
PROJECT_ROOT / "docs/examples/analytics/data/synthetic_medium_sa_assignments.csv",
),
(
"large",
PROJECT_ROOT / "examples/synthetic/large/scenario.yaml",
PROJECT_ROOT / "docs/examples/analytics/data/synthetic_large_sa_assignments.csv",
),
]
summary_rows = []
ensemble_tables = {}
for tier, scenario_path, assign_path in SCENARIOS:
config = SyntheticDatasetConfig(
name=f"synthetic-{tier}", tier=tier, num_blocks=8, num_days=12, num_machines=4
)
tables, sampling = utils.run_stochastic_summary(scenario_path, assign_path, tier=tier)
ensemble_tables[tier] = tables
production = tables.shift.groupby("sample_id")["production_units"].sum()
downtime = tables.shift.groupby("sample_id")["downtime_hours"].sum()
summary_rows.append(
{
"tier": tier,
"mean_production": production.mean(),
"std_production": production.std(),
"mean_downtime_hours": downtime.mean(),
"samples": tables.shift["sample_id"].nunique(),
}
)
summary_df = pd.DataFrame(summary_rows)
summary_df