""" Lab 3 - Percentiles and CDFs for PV inverter data Reads the ORIGINAL sheet exported from Excel directly (no manual pre-cleaning), using the same reading technique fixed for Labs 1 and 2. """ import pandas as pd import numpy as np import matplotlib.pyplot as plt FILE = "PV2013January.csv" def load_pv_sheet(path): """ Load one day's PV sheet exactly as Excel exports it: line 1: empty line 2: sensor names (TimeStamp / SENS0500... / WR11LV02...) line 3: the real column names (TimeStamp, ExlSolIrr, IntSolIrr, ...) line 4: units (hh:mm, W/m^2, W/m^2, ...) line 5+: data, with decimal numbers written as "11227,85" (comma instead of a dot, wrapped in quotes because a comma is otherwise the column separator). Returns a DataFrame with a proper TimeStamp column and every other column converted to a real number. """ df = pd.read_csv(path, skiprows=2) # ignore the 2 rows above the header df = df.drop(index=0).reset_index(drop=True) # drop the units row df = df.rename(columns={df.columns[0]: "TimeStamp"}) numeric_cols = df.columns.drop("TimeStamp") df[numeric_cols] = df[numeric_cols].apply( lambda col: pd.to_numeric(col.astype(str).str.replace(",", ".", regex=False), errors="coerce") ) return df df = load_pv_sheet(FILE) print("Loaded", len(df), "rows,", len(df.columns), "columns") print(df[["TimeStamp", "TmpMdul C", "Fac", "Pac"]].head()) def cdf_xy(data): """Return sorted values and their empirical CDF (percentile rank, 0-1).""" xs = np.sort(data.to_numpy()) n = len(xs) ys = np.arange(1, n + 1) / n return xs, ys def report_percentiles(name, data): p25, p50, p75 = np.percentile(data, [25, 50, 75]) print(f"{name}: n={len(data)} 25th={p25:.3f} 50th (median)={p50:.3f} 75th={p75:.3f}") return p25, p50, p75 # ===================================================================== # Column F: TmpMdul C (module temperature) # ===================================================================== tmod = df["TmpMdul C"] print("\n--- F: TmpMdul C ---") report_percentiles("TmpMdul C (raw)", tmod) print("zeros in TmpMdul C:", int((tmod == 0).sum())) fig, ax = plt.subplots(figsize=(6, 4.2)) xs, ys = cdf_xy(tmod) ax.plot(xs, ys, color="#2F3C7E") ax.set_xlabel("Module temperature, C") ax.set_ylabel("CDF") ax.set_title("F -- TmpMdul C: CDF (raw)") fig.tight_layout() fig.savefig("F_cdf_raw.png", dpi=150) plt.close(fig) # TmpMdul C has no physical zeros to filter (module temperature is never # exactly 0 C in this dataset), but we check anyway, per the general # "filter zeros if needed" instruction. tmod_nz = tmod[tmod != 0] if len(tmod_nz) != len(tmod): report_percentiles("TmpMdul C (zeros removed)", tmod_nz) fig, ax = plt.subplots(figsize=(6, 4.2)) xs, ys = cdf_xy(tmod_nz) ax.plot(xs, ys, color="#2F3C7E") ax.set_xlabel("Module temperature, C") ax.set_ylabel("CDF") ax.set_title("F -- TmpMdul C: CDF (zeros removed)") fig.tight_layout() fig.savefig("F_cdf_filtered.png", dpi=150) plt.close(fig) else: print("No zeros found in TmpMdul C -- filtered CDF is identical to the raw one.") # ===================================================================== # Column I: Fac (AC / grid frequency) -- dominated by zeros (inverter idle) # ===================================================================== fac = df["Fac"] print("\n--- I: Fac ---") report_percentiles("Fac (raw, incl. zeros)", fac) fac_zeros = int((fac == 0).sum()) print("zeros in Fac:", fac_zeros, f"({fac_zeros/len(fac)*100:.1f}% of readings)") fig, ax = plt.subplots(figsize=(6, 4.2)) xs, ys = cdf_xy(fac) ax.plot(xs, ys, color="#F96167") ax.set_xlabel("Fac, Hz") ax.set_ylabel("CDF") ax.set_title("I -- Fac: CDF (raw, incl. zeros)") fig.tight_layout() fig.savefig("I_cdf_raw.png", dpi=150) plt.close(fig) fac_nz = fac[fac > 0] report_percentiles("Fac (zeros removed)", fac_nz) fig, ax = plt.subplots(figsize=(6, 4.2)) xs, ys = cdf_xy(fac_nz) ax.plot(xs, ys, color="#F96167") ax.set_xlabel("Fac, Hz") ax.set_ylabel("CDF") ax.set_title("I -- Fac: CDF (zeros removed)") fig.tight_layout() fig.savefig("I_cdf_filtered.png", dpi=150) plt.close(fig) # ===================================================================== # Column P: Pac (AC power) -- also dominated by zeros (no generation at night) # ===================================================================== pac = df["Pac"] print("\n--- P: Pac ---") report_percentiles("Pac (raw, incl. zeros)", pac) pac_zeros = int((pac == 0).sum()) print("zeros in Pac:", pac_zeros, f"({pac_zeros/len(pac)*100:.1f}% of readings)") fig, ax = plt.subplots(figsize=(6, 4.2)) xs, ys = cdf_xy(pac) ax.plot(xs, ys, color="#2F3C7E") ax.set_xlabel("Pac, W") ax.set_ylabel("CDF") ax.set_title("P -- Pac: CDF (raw, incl. zeros)") fig.tight_layout() fig.savefig("P_cdf_raw.png", dpi=150) plt.close(fig) pac_nz = pac[pac > 0] report_percentiles("Pac (zeros removed)", pac_nz) fig, ax = plt.subplots(figsize=(6, 4.2)) xs, ys = cdf_xy(pac_nz) ax.plot(xs, ys, color="#2F3C7E") ax.set_xlabel("Pac, W") ax.set_ylabel("CDF") ax.set_title("P -- Pac: CDF (zeros removed)") fig.tight_layout() fig.savefig("P_cdf_filtered.png", dpi=150) plt.close(fig) print("\nAll figures saved.")