diff --git a/.github/find_missing_sources.py b/.github/find_missing_sources.py new file mode 100644 index 0000000..fd395f3 --- /dev/null +++ b/.github/find_missing_sources.py @@ -0,0 +1,143 @@ +#!/usr/bin/env python3 +""" +find_missing_source.py + +Place this file in: + root_folder/.github/find_missing_source.py + +It will automatically scan: + root_folder/Data + +and write output next to the script (in .github/) by default. +""" + +from __future__ import annotations + +import argparse +import os +from pathlib import Path +from typing import Optional, Tuple + +import pandas as pd + + +def detect_source_column(columns) -> Optional[str]: + """Return the actual column name whose stripped lowercase equals 'source'.""" + for c in columns: + if str(c).strip().lower() == "source": + return c + return None + + +def read_csv_robust(path: Path) -> Tuple[Optional[pd.DataFrame], Optional[str]]: + """Read CSV with delimiter sniffing and tolerant parsing.""" + try: + df = pd.read_csv( + path, + sep=None, # sniff delimiter + engine="python", # required for sep=None + dtype=str, # keep as strings + keep_default_na=True, + on_bad_lines="skip", + encoding_errors="replace", + ) + return df, None + except Exception as e: + return None, f"{type(e).__name__}: {e}" + + +def is_missing_source(series: pd.Series) -> pd.Series: + """True if value is NaN or only whitespace after stripping.""" + s = series.astype("string") + return s.isna() | (s.str.strip() == "") + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument( + "--out", + type=str, + default="missing_sources.xlsx", + help="Output file name (.csv or .xlsx). Default: missing_sources.xlsx", + ) + ap.add_argument( + "--include-errors", + action="store_true", + help="Also write a *_errors.csv listing unreadable CSVs and reasons.", + ) + args = ap.parse_args() + + # Script is in root_folder/.github/ + script_dir = Path(__file__).resolve().parent # root_folder/.github + root_folder = script_dir.parent # root_folder + data_dir = root_folder / "Data" # root_folder/Data + + if not data_dir.exists() or not data_dir.is_dir(): + raise SystemExit(f'Expected folder not found: "{data_dir}"') + + out_path = (script_dir / args.out).resolve() + + missing_rows = [] + errors = [] + + for dirpath, _, filenames in os.walk(data_dir): + for fn in filenames: + if not fn.lower().endswith(".csv"): + continue + + csv_path = Path(dirpath) / fn + df, err = read_csv_robust(csv_path) + if df is None: + errors.append({"file": str(csv_path), "error": err}) + continue + + source_col = detect_source_column(df.columns) + if source_col is None: + # No source column; skip silently (or record as error if you want) + continue + + mask = is_missing_source(df[source_col]) + if not mask.any(): + continue + + hit = df.loc[mask].copy() + + # Add traceability columns + # Store path relative to root_folder for cleaner output + rel_path = csv_path.relative_to(root_folder) + hit.insert(0, "__file__", str(rel_path)) + + # Row number in original file: +2 to account for header being row 1 + hit.insert(1, "__row__", (hit.index.to_series() + 2).astype(int)) + + missing_rows.append(hit) + + if missing_rows: + result = pd.concat(missing_rows, ignore_index=True) + else: + result = pd.DataFrame(columns=["__file__", "__row__"]) + + # Write main output + if out_path.suffix.lower() == ".xlsx": + with pd.ExcelWriter(out_path, engine="openpyxl") as writer: + result.to_excel(writer, index=False, sheet_name="missing_source") + else: + result.to_csv(out_path, index=False, encoding="utf-8") + + # Optionally write errors + if args.include_errors: + err_path = out_path.with_name(out_path.stem + "_errors.csv") + pd.DataFrame(errors).to_csv(err_path, index=False, encoding="utf-8") + + print(f"Script location: {script_dir}") + print(f"Scanning folder: {data_dir}") + print(f"Missing rows: {len(result)}") + print(f"Wrote output: {out_path}") + if args.include_errors: + print(f"Wrote errors: {err_path}") + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/.github/workflows/missing-source-check.yml b/.github/workflows/missing-source-check.yml new file mode 100644 index 0000000..ac964e2 --- /dev/null +++ b/.github/workflows/missing-source-check.yml @@ -0,0 +1,90 @@ +name: Missing Source Check + +on: + pull_request: + paths: + - "**/*.csv" # only run when CSVs change + +permissions: + contents: read + pull-requests: write # needed for PR comments + issues: write # comments are issued via the Issues API for PRs + +jobs: + missing-source: + runs-on: ubuntu-latest + + outputs: + missing_count: ${{ steps.count.outputs.missing_count }} + + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install pandas openpyxl + + - name: Run missing-source scan + run: | + python .github/find_missing_sources.py --out missing_sources.csv + + - name: Count missing rows + id: count + shell: bash + run: | + # Count data rows in CSV (minus header). If file is empty/nonexistent, count = 0. + if [ -f ".github/missing_sources.csv" ]; then + lines=$(wc -l < ".github/missing_sources.csv" | tr -d ' ') + if [ "$lines" -gt 0 ]; then + missing_count=$((lines - 1)) + else + missing_count=0 + fi + else + missing_count=0 + fi + + echo "missing_count=$missing_count" >> "$GITHUB_OUTPUT" + echo "Missing rows: $missing_count" >> "$GITHUB_STEP_SUMMARY" + + - name: Upload report artifact + uses: actions/upload-artifact@v4 + with: + name: missing-sources-report + path: .github/missing_sources.csv + if-no-files-found: warn + + # Optional: Fail the check if missing rows exist (uncomment if you want to block merges) + # - name: Fail if missing rows found + # if: ${{ steps.count.outputs.missing_count != '0' }} + # run: | + # echo "Found missing source rows: ${{ steps.count.outputs.missing_count }}" + # exit 1 + + - name: Comment on PR (skips forks) + if: ${{ github.event.pull_request.head.repo.fork == false }} + uses: actions/github-script@v7 + with: + script: | + const missing = Number("${{ steps.count.outputs.missing_count }}"); + const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`; + + const body = + `🧾 **Missing Source Check**\n\n` + + `Missing rows: **${missing}**\n\n` + + `Download the artifact (**missing-sources-report**) from the workflow run:\n` + + `${runUrl}`; + + await github.rest.issues.createComment({ + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: context.issue.number, + body + }); diff --git a/Data/Parameters/Par_AnnualMinNewCapacity/Europe_EnVis_Green/Par_AnnualMinNewCapacity.csv b/Data/Parameters/Par_AnnualMinNewCapacity/Europe_EnVis_Green/Par_AnnualMinNewCapacity.csv index 4ed0dc8..359ab67 100644 --- a/Data/Parameters/Par_AnnualMinNewCapacity/Europe_EnVis_Green/Par_AnnualMinNewCapacity.csv +++ b/Data/Parameters/Par_AnnualMinNewCapacity/Europe_EnVis_Green/Par_AnnualMinNewCapacity.csv @@ -17,3 +17,4 @@ UK,P_Nuclear,2040,1.631,,GW,Partner Feedback Man0EUvRE - Sandrine Charousset,03. UK,P_Nuclear,2045,1.632,,GW,Partner Feedback Man0EUvRE - Sandrine Charousset,03.06.2025,Nikita Moskalenko TR,P_Nuclear,2025,1.2,,GW,Partner Feedback Man0EUvRE - Sandrine Charousset,03.06.2025,Nikita Moskalenko TR,P_Nuclear,2030,3.6,,GW,Partner Feedback Man0EUvRE - Sandrine Charousset,03.06.2025,Nikita Moskalenko +TR,TEST,All,0.1,,GW,,,Test