diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..cb9b417 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,50 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + +jobs: + test: + runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + python-version: ["3.9", "3.10", "3.11", "3.12"] + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + cache: pip + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -r requirements-dev.txt + + - name: Run tests + run: pytest tests.py -v + + audit: + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.12" + cache: pip + + - name: Install pip-audit + run: pip install pip-audit + + - name: Audit declared dependencies for known CVEs + run: pip-audit -r requirements.txt --strict diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..8a672d4 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Shawn P (MsShawnP) + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md index 8092d53..e99a9fd 100644 --- a/README.md +++ b/README.md @@ -42,11 +42,17 @@ This tool gives you a clear picture in 60 seconds. ## Run locally +Requires Python 3.9+. + ```bash # Clone the repo -git clone https://github.com/YOUR_USERNAME/gtin-validator.git +git clone https://github.com/MsShawnP/gtin-validator.git cd gtin-validator +# (Optional) create a virtual environment +python -m venv .venv +source .venv/bin/activate # Windows: .venv\Scripts\activate + # Install dependencies pip install -r requirements.txt @@ -54,6 +60,13 @@ pip install -r requirements.txt streamlit run app.py ``` +## Run tests + +```bash +pip install -r requirements-dev.txt +pytest tests.py -v +``` + ## Deploy to Streamlit Community Cloud 1. Push this repo to GitHub @@ -68,14 +81,17 @@ streamlit run app.py ``` gtin-validator/ -├── app.py # Streamlit UI -├── gtin_core.py # Validation engine, scoring, retailer rules -├── csv_report.py # CSV export -├── pdf_report.py # Branded PDF report (reportlab) -├── sample_data.py # Realistic sample data -├── requirements.txt +├── app.py # Streamlit UI +├── gtin_core.py # Validation engine, scoring, retailer rules +├── csv_report.py # CSV export +├── pdf_report.py # Branded PDF report (reportlab) +├── sample_data.py # Realistic sample data +├── tests.py # pytest suite +├── requirements.txt # Runtime dependencies +├── requirements-dev.txt # Test/development dependencies +├── LICENSE # MIT ├── .streamlit/ -│ └── config.toml # Theme configuration +│ └── config.toml # Theme configuration └── README.md ``` @@ -102,4 +118,4 @@ MIT --- -*Built as a portfolio piece demonstrating product data consulting for specialty food brands. For a comprehensive Product Data Health Audit for your brand, [get in touch](mailto:YOUR_EMAIL).* +*Built as a portfolio piece demonstrating product data consulting for specialty food brands. For a comprehensive Product Data Health Audit for your brand, [get in touch](mailto:Shawn@lailarallc.com).* diff --git a/app.py b/app.py index 343615b..d4f5372 100644 --- a/app.py +++ b/app.py @@ -4,20 +4,28 @@ for retailer submission (Walmart, Costco, UNFI, 1WorldSync, and more). Built for operations people, not developers. + +This file is the Streamlit entry point — it wires together page config, +the sidebar, and the rendering modules under ui/. The substantive UI +lives there. """ import streamlit as st -import pandas as pd -from io import StringIO -from gtin_core import ( - validate_batch, Severity, generate_before_after, - RETAILER_PROFILES, GTINType, - generate_executive_summary, generate_fix_roadmap, - generate_gtin14_suggestions, check_data_completeness, + +from gtin_core import RETAILER_PROFILES, validate_batch +from ui.deep_analysis import render_deep_analysis +from ui.input_section import render_input_section +from ui.results import ( + render_download_buttons, + render_results_tabs, + render_score_card, + render_summary_stats, +) +from ui.state import ( + KEY_DF, KEY_GTINS, KEY_VALIDATED, KEY_VALIDATION_CACHE, + invalidate_report_caches, reset_session, ) -from csv_report import generate_csv_report -from pdf_report import generate_pdf_report -from sample_data import SAMPLE_DATA, SAMPLE_DESCRIPTION +from ui.styles import inject_css # --------------------------------------------------------------------------- @@ -31,201 +39,7 @@ initial_sidebar_state="expanded", ) -# --------------------------------------------------------------------------- -# Custom CSS -# --------------------------------------------------------------------------- - -theme_css = """ -:root { - --bg-primary: #eaecee; - --bg-secondary: #e0e2e5; - --bg-card: #f5f5f5; - --bg-input: #ffffff; - --text-primary: #1a1a2e; - --text-secondary: #4a4a5a; - --text-muted: #6c757d; - --border-color: #d0d3d8; - --stat-card-bg: #f0f1f3; - --stat-card-border: #d0d3d8; - --retailer-card-bg: #f5f5f5; - --cost-card-bg: linear-gradient(135deg, #fff3cd 0%, #ffeeba 100%); - --cost-card-border: #ffc107; - --cost-number-color: #856404; - --security-bg: #e8f5e9; - --security-border: #c3e6cb; -} -.stApp { background-color: #eaecee !important; } -[data-testid="stSidebar"] { background-color: #e0e2e5 !important; } -.stTabs [data-baseweb="tab"] { color: #1a1a2e !important; } -.stTabs [data-baseweb="tab"][aria-selected="true"] { - color: #1a1a2e !important; - font-weight: 600; -} -.stTextInput input, .stTextArea textarea { - background-color: #ffffff !important; - color: #1a1a2e !important; - border-color: #d0d3d8 !important; -} -[data-baseweb="select"], -[data-baseweb="select"] div, -[data-baseweb="select"] span { - color: #1a1a2e !important; -} -""" - -st.markdown(f""" - -""", unsafe_allow_html=True) +inject_css() # --------------------------------------------------------------------------- @@ -238,6 +52,7 @@ "Built for operations teams at specialty food brands preparing for national retail." ) + # --------------------------------------------------------------------------- # Sidebar — input + settings # --------------------------------------------------------------------------- @@ -253,14 +68,12 @@ st.markdown("---") - # Company name for branding company_name = st.text_input( "Your company name (optional)", placeholder="e.g., Cedar Hollow Provisions", help="Used to brand your PDF report.", ) - # Retailer filter st.markdown("### Filter by retailer") selected_retailer = st.selectbox( "Show requirements for:", @@ -284,645 +97,46 @@ st.markdown("---") -input_method = st.radio( - "Choose input method:", - ["Paste GTINs", "Upload CSV", "Try sample data"], - horizontal=True, -) - -gtins_to_validate = [] -uploaded_df = None # Store full DataFrame for data completeness check - -if input_method == "Paste GTINs": - gtin_input = st.text_area( - "Paste your GTINs (one per line):", - height=200, - placeholder="614141000012\n614141000029\n614141000036\n...", - ) - if gtin_input.strip(): - gtins_to_validate = [ - line.strip() for line in gtin_input.strip().split("\n") - if line.strip() - ] - -elif input_method == "Upload CSV": - uploaded_file = st.file_uploader( - "Upload a CSV file with a GTIN column:", - type=["csv"], - help="Your CSV should have a column containing GTINs. We'll auto-detect it.", - ) - if uploaded_file: - try: - df = pd.read_csv(uploaded_file, dtype=str) - uploaded_df = df # Save for data completeness - # Auto-detect GTIN column - gtin_col = None - for col in df.columns: - if any(term in col.lower() for term in ["gtin", "upc", "ean", "barcode", "code"]): - gtin_col = col - break - if gtin_col is None: - gtin_col = st.selectbox( - "Which column contains GTINs?", - df.columns.tolist(), - ) - else: - st.info(f"Auto-detected GTIN column: **{gtin_col}**") - - gtins_to_validate = df[gtin_col].dropna().tolist() - st.success(f"Loaded {len(gtins_to_validate)} GTINs from '{gtin_col}'") - except Exception as e: - st.error(f"Error reading CSV: {e}") - -elif input_method == "Try sample data": - st.markdown(SAMPLE_DESCRIPTION) - sample_df = pd.read_csv(StringIO(SAMPLE_DATA.strip()), dtype=str) - uploaded_df = sample_df # Save for data completeness - st.dataframe(sample_df, use_container_width=True, height=300) - gtins_to_validate = sample_df["GTIN"].dropna().tolist() - st.info(f"Loaded {len(gtins_to_validate)} sample GTINs") - - -# --------------------------------------------------------------------------- -# Validation -# --------------------------------------------------------------------------- +gtins_to_validate, uploaded_df = render_input_section() if gtins_to_validate: - # Store GTINs in session state so dark mode toggle doesn't lose them - st.session_state["gtins_to_validate"] = gtins_to_validate + # Persist parsed input so it survives Streamlit reruns triggered by + # unrelated widget interactions. + st.session_state[KEY_GTINS] = gtins_to_validate if uploaded_df is not None: - st.session_state["uploaded_df"] = uploaded_df - -# Recover from session state if input was lost (e.g., after dark mode toggle) -if not gtins_to_validate and st.session_state.get("validated") and st.session_state.get("gtins_to_validate"): - gtins_to_validate = st.session_state["gtins_to_validate"] - uploaded_df = st.session_state.get("uploaded_df", uploaded_df) + st.session_state[KEY_DF] = uploaded_df -if gtins_to_validate: - btn_col1, btn_col2 = st.columns([3, 1]) - with btn_col1: - validate_btn = st.button("🔍 Validate GTINs", type="primary", use_container_width=True) - with btn_col2: - reset_btn = st.button("🔄 Reset", use_container_width=True) - - if reset_btn: - for key in list(st.session_state.keys()): - del st.session_state[key] - st.rerun() - - if validate_btn or st.session_state.get("validated"): - st.session_state["validated"] = True - - # Use cached validation data if available, otherwise validate - if validate_btn or "validation_data_cache" not in st.session_state: - with st.spinner("Validating your GTINs against GS1 standards..."): - validation_data = validate_batch(gtins_to_validate) - st.session_state["validation_data_cache"] = validation_data - else: - validation_data = st.session_state["validation_data_cache"] - - summary = validation_data["summary"] - score = validation_data["score"] - cost = validation_data["cost_estimate"] - results = validation_data["results"] - hierarchy = validation_data["hierarchy"] - retailer_checklists = validation_data["retailer_checklists"] - - st.markdown("---") - - # === READINESS SCORE === - score_color = "#28a745" if score["score"] >= 75 else ("#ffc107" if score["score"] >= 50 else "#dc3545") - st.markdown(f""" -
-
{score["score"]}
-
Grade: {score["grade"]}
-
{score["interpretation"]}
-
- """, unsafe_allow_html=True) - - # === SUMMARY STATS === - col1, col2, col3, col4 = st.columns(4) - with col1: - st.markdown(f""" -
-
{summary["total_gtins"]}
-
Total GTINs
-
- """, unsafe_allow_html=True) - with col2: - st.markdown(f""" -
-
{summary["critical_issues"]}
-
Critical Issues
-
- """, unsafe_allow_html=True) - with col3: - st.markdown(f""" -
-
{summary["warnings"]}
-
Warnings
-
- """, unsafe_allow_html=True) - with col4: - st.markdown(f""" -
-
{summary["clean"]}
-
Clean
-
- """, unsafe_allow_html=True) - - # === DOWNLOAD VALIDATION REPORTS (above results) === - st.markdown("### 📥 Download Validation Reports") - - dl_col1, dl_col2 = st.columns(2) - - with dl_col1: - st.markdown("**📄 CSV Report — Raw Data**") - st.markdown( - '

' - 'Row-by-row validation results in spreadsheet format. ' - 'Includes each GTIN, its status, issue codes, and corrected values. ' - 'Best for importing into Excel or your own systems for further analysis.' - '

', - unsafe_allow_html=True, - ) - csv_data = generate_csv_report(validation_data) - filename_base = company_name.replace(" ", "_") if company_name else "gtin_validation" - st.download_button( - label="📄 Download CSV Report", - data=csv_data, - file_name=f"{filename_base}_report.csv", - mime="text/csv", - use_container_width=True, - ) - - with dl_col2: - st.markdown("**📑 PDF Report — Full Diagnostic**") - st.markdown( - '

' - 'Branded, professional report with readiness score, retailer-specific ' - 'checklists, cost-of-inaction estimates, and prioritized issue detail. ' - 'Designed to hand directly to your operations team, broker, or trading partner coordinator.' - '

', - unsafe_allow_html=True, - ) - try: - pdf_buffer = generate_pdf_report(validation_data, company_name) - st.download_button( - label="📑 Download PDF Report", - data=pdf_buffer, - file_name=f"{filename_base}_report.pdf", - mime="application/pdf", - use_container_width=True, - ) - except Exception as e: - st.error(f"PDF generation error: {e}") - - # === VALIDATION RESULTS TABS === - st.markdown("---") - st.markdown("### Validation Results") - - tab_issues, tab_detail, tab_check_digit_fixes, tab_item_detail = st.tabs([ - "📋 Issues by Severity", - "🔍 Full Item Detail", - "✏️ Check Digit Corrections", - "📦 Packaging Hierarchy", - ]) - - # --- Issues by Severity --- - with tab_issues: - st.markdown("### Issues by Severity") - - critical_items = [r for r in results if r.has_critical] - warning_items = [r for r in results if r.has_warning and not r.has_critical] - info_items = [r for r in results if r.issues and not r.has_critical and not r.has_warning] - - if critical_items: - st.markdown(f'CRITICAL — ' - f'These GTINs will be **rejected** by retailers.', - unsafe_allow_html=True) - for r in critical_items: - with st.expander(f"Row {r.row_number}: {r.raw_input}"): - for issue in r.issues: - if issue.severity == Severity.CRITICAL: - st.error(f"**{issue.message}**") - st.markdown(f"**Fix:** {issue.recommendation}") - st.markdown(f"**Retailer impact:** {issue.retailer_impact}") - st.markdown("---") - - if warning_items: - st.markdown(f'WARNING — ' - f'These GTINs may cause problems.', - unsafe_allow_html=True) - for r in warning_items: - with st.expander(f"Row {r.row_number}: {r.raw_input}"): - for issue in r.issues: - st.warning(f"**{issue.message}**") - st.markdown(f"**Fix:** {issue.recommendation}") - st.markdown(f"**Retailer impact:** {issue.retailer_impact}") - - if info_items: - st.markdown(f'INFO — ' - f'Best practice notes.', - unsafe_allow_html=True) - for r in info_items: - with st.expander(f"Row {r.row_number}: {r.raw_input}"): - for issue in r.issues: - st.info(f"{issue.message}") - - if not critical_items and not warning_items and not info_items: - st.success("🎉 All GTINs passed validation with no issues!") - - # --- Check Digit Corrections --- - with tab_check_digit_fixes: - st.markdown("### Check Digit Corrections") - st.markdown( - "These GTINs have incorrect check digits. The corrected values are shown below. " - "**Important:** always verify corrections against your original barcode or GS1 " - "registration before updating your product master." - ) - - before_after = generate_before_after(results) - if before_after: - ba_df = pd.DataFrame(before_after) - ba_df.columns = ["Row", "Current (Before)", "Corrected (After)", "Issue"] - st.dataframe(ba_df, use_container_width=True, hide_index=True) - else: - st.success("No check digit corrections needed — all check digits are valid.") - - # --- Packaging Hierarchy --- - with tab_item_detail: - st.markdown("### Packaging Hierarchy Analysis") - st.markdown( - "Retailers like Walmart require GTINs at every packaging level — " - "each, inner pack, case, and pallet. This analysis checks whether " - "your case-level GTIN-14s match up with unit-level GTINs." - ) - - if hierarchy["matched_pairs"]: - st.markdown("#### ✅ Matched unit → case pairs") - pairs_df = pd.DataFrame(hierarchy["matched_pairs"]) - pairs_df.columns = ["Case GTIN", "Case Row", "Unit GTIN", "Unit Row", "Indicator"] - st.dataframe(pairs_df, use_container_width=True, hide_index=True) - - if hierarchy["orphan_cases"]: - st.markdown("#### ⚠️ Case GTINs without matching unit GTINs") - for r in hierarchy["orphan_cases"]: - st.warning(f"Row {r.row_number}: **{r.cleaned}** — no matching unit GTIN found") - - if hierarchy["units_without_cases"]: - st.markdown("#### 📦 Unit GTINs without case-level GTINs") - st.caption( - "These items don't have a corresponding GTIN-14 for case/shipping identification. " - "If you ship these to retailers in cases, you'll need case GTINs." - ) - for r in hierarchy["units_without_cases"]: - st.info(f"Row {r.row_number}: **{r.cleaned}** ({r.gtin_type.value})") - - if not hierarchy["matched_pairs"] and not hierarchy["orphan_cases"]: - st.info( - "No GTIN-14 case-level codes found in your data. " - "If you ship to Walmart or Costco, you'll likely need case GTINs (GTIN-14 with indicator digits 1-8)." - ) - - # --- Full Item Detail --- - with tab_detail: - st.markdown("### Full Item Detail") - - detail_rows = [] - for r in results: - status = "✅ Clean" if not r.issues else ( - "❌ Critical" if r.has_critical else ( - "⚠️ Warning" if r.has_warning else "ℹ️ Info" - ) - ) - detail_rows.append({ - "Row": r.row_number, - "GTIN": r.raw_input, - "Type": r.gtin_type.value, - "Status": status, - "Issues": len(r.issues), - "Corrected": r.corrected_value or "", - }) - - detail_df = pd.DataFrame(detail_rows) - st.dataframe(detail_df, use_container_width=True, hide_index=True) - - # ================================================================= - # DEEP ANALYSIS SECTION - # ================================================================= - st.markdown("---") - st.markdown("## 🔬 Deep Analysis") - st.markdown( - "Go beyond basic validation — understand what to fix first, " - "check retailer readiness, estimate costs, and track your progress." - ) +# Recover the parsed input from session state when the current rerun lost it. +if ( + not gtins_to_validate + and st.session_state.get(KEY_VALIDATED) + and st.session_state.get(KEY_GTINS) +): + gtins_to_validate = st.session_state[KEY_GTINS] + uploaded_df = st.session_state.get(KEY_DF, uploaded_df) - tab_summary, tab_roadmap, tab_retailer, tab_cost, tab_gtin14, tab_completeness = st.tabs([ - "📝 Executive Summary", - "🗺️ Prioritized Fix Plan", - "🏪 Retailer Readiness", - "💰 Cost of Inaction", - "🔧 Case GTIN Generator", - "📊 Product Data Completeness", - ]) - - # --- Executive Summary --- - with tab_summary: - st.markdown("### Executive Summary") - st.markdown( - "Copy this summary and send it to your team. " - "It's written in plain language — no jargon." - ) - - exec_summary = generate_executive_summary(validation_data) - st.markdown( - f'
' - f'{exec_summary.replace(chr(10)+chr(10), "

")}' - f'
', - unsafe_allow_html=True, - ) - - st.download_button( - label="📋 Copy Summary as Text", - data=exec_summary, - file_name="gtin_executive_summary.txt", - mime="text/plain", - use_container_width=True, - ) - - # --- Prioritized Fix Plan --- - with tab_roadmap: - st.markdown("### Prioritized Fix Plan") - st.markdown( - "Issues ranked by **impact × effort**. Start at the top — " - "these are your fastest wins with the biggest payoff." - ) - - roadmap = generate_fix_roadmap(results, hierarchy) - if roadmap: - for idx, item in enumerate(roadmap, 1): - effort_color = {"Low": "#28a745", "Medium": "#ffc107", "High": "#dc3545"}.get(item["effort"], "#6c757d") - impact_color = {"High": "#dc3545", "Medium": "#ffc107", "Low": "#28a745"}.get(item["impact"], "#6c757d") - - with st.expander( - f"Priority {idx}: {item['action'][:80]}{'...' if len(item['action']) > 80 else ''} " - f"({item['count']} item{'s' if item['count'] != 1 else ''})" - ): - col_e, col_i, col_t = st.columns(3) - with col_e: - st.markdown(f"**Effort:** {item['effort']}", - unsafe_allow_html=True) - st.caption(item["effort_detail"]) - with col_i: - st.markdown(f"**Impact:** {item['impact']}", - unsafe_allow_html=True) - st.caption(item["impact_detail"]) - with col_t: - st.markdown(f"**Time estimate:**") - st.caption(item["time_estimate"]) - - st.markdown(f"**Full recommendation:** {item['action']}") - else: - st.success("No issues to fix — your data is clean!") - - # --- Retailer Readiness --- - with tab_retailer: - st.markdown("### Retailer Submission Readiness") - st.markdown( - "Each retailer has specific GTIN requirements. " - "Here's how your data stacks up." - ) - - retailers_to_show = ( - {selected_retailer: retailer_checklists[selected_retailer]} - if selected_retailer != "All Retailers" - else retailer_checklists - ) - - for retailer_name, checklist in retailers_to_show.items(): - ready_class = "retailer-ready" if checklist["ready"] else "retailer-not-ready" - status_text = "✅ READY" if checklist["ready"] else f"❌ {checklist['passed']}/{checklist['total']} checks passed" - - st.markdown(f""" -
- {retailer_name} — {status_text}
- {checklist['profile']['description']} -
- """, unsafe_allow_html=True) - - for check in checklist["checks"]: - icon = "✅" if check["passed"] else "❌" - st.markdown(f"    {icon} {check['check']} — *{check['detail']}*") - - if checklist["profile"].get("notes"): - st.caption(checklist["profile"]["notes"]) - - st.markdown("") - - # --- Cost of Inaction --- - with tab_cost: - st.markdown("### Estimated Cost of Inaction") - st.markdown( - "These estimates are based on industry averages for specialty food brands " - "at similar scale. They're directional — meant to quantify the risk, not " - "predict exact costs." - ) - - if cost: - col_a, col_b = st.columns(2) - with col_a: - st.markdown(f""" -
-
- ${cost['annual_estimate_low']:,} – ${cost['annual_estimate_high']:,} -
-
Estimated annual cost of unresolved GTIN issues
-
- """, unsafe_allow_html=True) - - with col_b: - st.markdown(f""" -
-
{cost['rework_hours']} hours/year
-
Manual rework from GTIN problems
-
- """, unsafe_allow_html=True) - - st.markdown("#### Breakdown") - cost_df = pd.DataFrame([ - { - "Category": "Chargebacks from invalid GTINs", - "Low Estimate": f"${cost['chargeback_range'][0]:,}", - "High Estimate": f"${cost['chargeback_range'][1]:,}", - }, - { - "Category": f"Delayed launches ({cost['delayed_skus']} SKUs)", - "Low Estimate": f"${cost['delayed_launch_range'][0]:,}", - "High Estimate": f"${cost['delayed_launch_range'][1]:,}", - }, - { - "Category": f"Manual rework ({cost['rework_hours']} hrs)", - "Low Estimate": f"${cost['rework_cost']:,}", - "High Estimate": f"${cost['rework_cost']:,}", - }, - ]) - st.dataframe(cost_df, use_container_width=True, hide_index=True) - - if cost.get("growth_note"): - st.warning(f"📈 **Growth multiplier:** {cost['growth_note']}") - else: - st.info("No cost estimates available — no issues detected.") - - # --- Case GTIN Generator --- - with tab_gtin14: - st.markdown("### Case GTIN-14 Generator") - st.markdown( - "These are your unit-level GTINs that don't have a corresponding " - "case-level GTIN-14 in your file. Below are the GTIN-14s you'd need " - "to create for each packaging level." - ) - - suggestions = generate_gtin14_suggestions(results, hierarchy) - if suggestions: - st.markdown( - f"**{len(suggestions)} unit GTIN(s)** need case-level GTIN-14s." - ) - - for s in suggestions: - with st.expander(f"Row {s['row']}: {s['unit_gtin']} ({s['unit_type']})"): - gtin14_rows = [] - for ind, info in s["indicators"].items(): - gtin14_rows.append({ - "Indicator": str(ind), - "GTIN-14": info["gtin14"], - "Packaging Level": info["label"], - }) - st.dataframe( - pd.DataFrame(gtin14_rows), - use_container_width=True, - hide_index=True, - ) - st.caption( - "Most commonly, indicator 1 = case. Copy the GTIN-14 you need " - "and add it to your product master." - ) - else: - st.success( - "All unit GTINs have matching case-level GTIN-14s, " - "or no valid unit GTINs were found to generate suggestions for." - ) - - # --- Product Data Completeness --- - with tab_completeness: - st.markdown("### Product Data Completeness") - - if uploaded_df is not None and len(uploaded_df.columns) > 1: - st.markdown( - "Beyond GTINs, retailers require dozens of product attributes. " - "Here's what we found in your file." - ) - - completeness = check_data_completeness(uploaded_df) - - if completeness["field_analysis"]: - overall = completeness["overall_completeness"] - overall_color = "#28a745" if overall >= 80 else ("#ffc107" if overall >= 50 else "#dc3545") - st.markdown( - f'
' - f'
' - f'{overall}%
' - f'
Overall Data Completeness
', - unsafe_allow_html=True, - ) - - st.markdown("#### Fields Found in Your File") - field_rows = [] - for field_name, data in completeness["field_analysis"].items(): - field_rows.append({ - "Field": field_name.replace("_", " ").title(), - "Column": data["column_name"], - "Populated": f"{data['populated']}/{data['total_rows']}", - "% Rows Populated": f"{data['completeness_pct']}%", - }) - st.dataframe(pd.DataFrame(field_rows), use_container_width=True, hide_index=True) - - if completeness["missing_important_fields"]: - st.markdown("#### Missing Important Fields") - st.warning( - "The following fields were not found in your file: **" + - ", ".join(f.replace("_", " ").title() for f in completeness["missing_important_fields"]) + - "**. Most retailers require these for item setup." - ) - - st.markdown("#### Retailer Data Readiness") - for retailer, gaps in completeness["retailer_data_gaps"].items(): - status = "✅ READY" if gaps["ready"] else f"❌ {gaps['present']}/{gaps['required']} fields present" - with st.expander(f"{retailer} — {status}"): - if gaps["missing_fields"]: - st.markdown( - "**Missing:** " + - ", ".join(f.replace("_", " ").title() for f in gaps["missing_fields"]) - ) - if gaps["incomplete_fields"]: - st.markdown( - "**Incomplete (not all rows filled):** " + - ", ".join(f.replace("_", " ").title() for f in gaps["incomplete_fields"]) - ) - if gaps["ready"]: - st.success("All required fields present and complete.") - else: - st.info( - "No standard product data fields detected beyond GTINs. " - "Upload a CSV with columns like Product Name, Brand, Weight, " - "Height, Width, Depth, etc. for a completeness analysis." - ) - else: - st.info( - "Data completeness analysis is available when you upload a CSV file " - "with multiple columns (beyond just GTINs). Upload a product master " - "spreadsheet to see which fields are missing or incomplete." - ) - - # === SHARE & SECURITY === - st.markdown("---") - st.markdown("### 🔗 Share Results") - st.info( - "To share these results, download the PDF report and send it to your team. " - "The branded report is designed to be forwarded to your operations team, broker, or " - "trading partner coordinator." - ) - st.markdown( - '
' - '🔒 Your Data Stays Yours — ' - 'No product data is stored, logged, or transmitted to third parties. ' - 'Everything is processed in-session and discarded when you close this page.' - '
', - unsafe_allow_html=True, - ) +# --------------------------------------------------------------------------- +# Validation flow +# --------------------------------------------------------------------------- -else: - # No data loaded yet — show explainer +def _no_data_explainer() -> None: st.markdown("---") st.markdown("### What this tool checks") col_a, col_b, col_c = st.columns(3) - with col_a: st.markdown("#### 🔢 Format & Structure") st.markdown( "Valid GTIN lengths (8, 12, 13, 14 digits), numeric-only, " "correct check digits using GS1's mod-10 algorithm." ) - with col_b: st.markdown("#### 🏪 Retailer Requirements") st.markdown( "Walmart Item 360, Costco, UNFI, KeHE, Whole Foods, " "1WorldSync — each has specific GTIN format and hierarchy requirements." ) - with col_c: st.markdown("#### 📦 Packaging Hierarchy") st.markdown( @@ -940,8 +154,67 @@ 'Your data is never used for training, analytics, or any purpose beyond generating ' 'your validation results in this session. When you close the tab, your data is gone.

' 'This tool runs on Streamlit Community Cloud. ' - 'Streamlit\'s infrastructure processes the request but does not persist application data between sessions. ' + "Streamlit's infrastructure processes the request but does not persist application data between sessions. " 'For details, see Streamlit\'s privacy policy.' '', unsafe_allow_html=True, ) + + +def _share_and_security_footer() -> None: + st.markdown("---") + st.markdown("### 🔗 Share Results") + st.info( + "To share these results, download the PDF report and send it to your team. " + "The branded report is designed to be forwarded to your operations team, broker, or " + "trading partner coordinator." + ) + st.markdown( + '
' + '🔒 Your Data Stays Yours — ' + 'No product data is stored, logged, or transmitted to third parties. ' + 'Everything is processed in-session and discarded when you close this page.' + '
', + unsafe_allow_html=True, + ) + + +if not gtins_to_validate: + _no_data_explainer() +else: + btn_col1, btn_col2 = st.columns([3, 1]) + with btn_col1: + validate_btn = st.button( + "🔍 Validate GTINs", type="primary", use_container_width=True, + ) + with btn_col2: + reset_btn = st.button("🔄 Reset", use_container_width=True) + + if reset_btn: + reset_session() + st.rerun() + + if validate_btn or st.session_state.get(KEY_VALIDATED): + st.session_state[KEY_VALIDATED] = True + + # Use cached validation data when nothing has changed; on a fresh + # validate run, also drop the derived report caches. + if validate_btn or KEY_VALIDATION_CACHE not in st.session_state: + with st.spinner("Validating your GTINs against GS1 standards..."): + validation_data = validate_batch(gtins_to_validate) + st.session_state[KEY_VALIDATION_CACHE] = validation_data + invalidate_report_caches() + else: + validation_data = st.session_state[KEY_VALIDATION_CACHE] + + st.markdown("---") + render_score_card(validation_data["score"]) + render_summary_stats(validation_data["summary"]) + render_download_buttons(validation_data, company_name) + + st.markdown("---") + st.markdown("### Validation Results") + render_results_tabs(validation_data) + + render_deep_analysis(validation_data, selected_retailer, uploaded_df) + _share_and_security_footer() diff --git a/csv_report.py b/csv_report.py index 9eefd7d..424cc67 100644 --- a/csv_report.py +++ b/csv_report.py @@ -8,6 +8,19 @@ from gtin_core import Severity +# Characters that trigger formula evaluation when a CSV is opened in +# Excel, LibreOffice Calc, or Google Sheets. Prefixing cells that start +# with one of these with a leading apostrophe neutralizes the formula. +_FORMULA_PREFIXES = ("=", "+", "-", "@", "\t", "\r") + + +def _sanitize_cell(value): + """Neutralize CSV/spreadsheet formula injection on user-controlled cells.""" + if isinstance(value, str) and value and value[0] in _FORMULA_PREFIXES: + return "'" + value + return value + + def generate_csv_report(validation_data: dict) -> str: """Generate a CSV report string from validation results.""" output = StringIO() @@ -57,18 +70,18 @@ def generate_csv_report(validation_data: dict) -> str: writer.writerow([ r.row_number, - r.raw_input, - r.cleaned, + _sanitize_cell(r.raw_input), + _sanitize_cell(r.cleaned), "Yes" if r.is_valid else "No", r.gtin_type.value, highest_severity, len(r.issues), - issues_text, - recommendations_text, - impact_text, - r.corrected_value or "", - r.company_prefix or "", - r.indicator_digit or "", + _sanitize_cell(issues_text), + _sanitize_cell(recommendations_text), + _sanitize_cell(impact_text), + _sanitize_cell(r.corrected_value or ""), + _sanitize_cell(r.company_prefix or ""), + _sanitize_cell(r.indicator_digit or ""), ]) return output.getvalue() diff --git a/gtin_core.py b/gtin_core.py index 8a470e9..367df94 100644 --- a/gtin_core.py +++ b/gtin_core.py @@ -128,7 +128,6 @@ def identify_gtin_type(length: int) -> GTINType: "case, pallet). All GTINs are validated against the GS1 database. " "Items with invalid GTINs will not go live in Item 360." ), - "gtin14_format": "GTIN-14 preferred for case-level identification", }, "Costco": { "description": "Costco Item Setup Workbook", @@ -139,7 +138,6 @@ def identify_gtin_type(length: int) -> GTINType: "Costco requires valid GTINs for all items. Dimension and weight " "discrepancies tied to wrong GTINs result in logistics chargebacks." ), - "gtin14_format": "GTIN-14 required for case/pallet levels", }, "UNFI": { "description": "UNFI New Item Form", @@ -150,7 +148,6 @@ def identify_gtin_type(length: int) -> GTINType: "UNFI requires UPC for each sellable unit. Case GTIN needed for " "warehouse receiving. Incorrect GTINs delay item activation." ), - "gtin14_format": "Case GTIN required for distribution", }, "Whole Foods": { "description": "Whole Foods Market Item Setup", @@ -161,7 +158,6 @@ def identify_gtin_type(length: int) -> GTINType: "Whole Foods requires valid UPC/EAN for each sellable unit. Items " "synced via 1WorldSync must have complete, accurate data." ), - "gtin14_format": "Not typically required at store level", }, "KeHE": { "description": "KeHE Distributors Item Setup", @@ -172,7 +168,6 @@ def identify_gtin_type(length: int) -> GTINType: "KeHE requires UPC for each sellable unit and case GTIN for " "warehouse operations. Data synced via 1WorldSync." ), - "gtin14_format": "Case GTIN required for distribution", }, "1WorldSync (GDSN)": { "description": "1WorldSync Global Data Synchronisation Network", @@ -185,7 +180,6 @@ def identify_gtin_type(length: int) -> GTINType: "configuration errors and logistics chargebacks. Wrong nutritional " "data creates legal exposure." ), - "gtin14_format": "Full hierarchy with indicator digits required", }, } @@ -194,7 +188,7 @@ def identify_gtin_type(length: int) -> GTINType: # Single-GTIN validation # ============================================================================= -def validate_single_gtin(raw: str, row_number: int) -> GTINResult: +def validate_single_gtin(raw, row_number: int) -> GTINResult: """ Validate a single GTIN string against GS1 standards. @@ -208,12 +202,25 @@ def validate_single_gtin(raw: str, row_number: int) -> GTINResult: 7. UPC-A → GTIN-13 format advisory Args: - raw: The raw GTIN string as entered by the user. + raw: The raw GTIN value as entered by the user (string preferred; + None and pandas NaN are coerced to empty for graceful handling). row_number: 1-based row position in the input file. Returns: A GTINResult with all issues found. """ + # Defensively coerce None / pandas NaN / non-string inputs to a string + # so we don't crash on .strip() when called from a DataFrame-driven flow. + if raw is None: + raw = "" + elif not isinstance(raw, str): + try: + if pd.isna(raw): + raw = "" + else: + raw = str(raw) + except (TypeError, ValueError): + raw = str(raw) cleaned = raw.strip().replace("-", "").replace(" ", "") result = GTINResult( raw_input=raw.strip(), @@ -368,7 +375,12 @@ def validate_single_gtin(raw: str, row_number: int) -> GTINResult: )) # --- Extract company prefix (approximate — real prefix length varies 7-10) --- - if gtin_type in (GTINType.GTIN_12, GTINType.GTIN_13): + # Normalize GTIN-12 and GTIN-14 onto the same GTIN-13 frame before + # slicing, so a unit/case pair (e.g. 614141000012 / 10614141000019) + # reports the same prefix and doesn't trip PREFIX_MISMATCH. + if gtin_type == GTINType.GTIN_12: + result.company_prefix = ("0" + cleaned)[:7] + elif gtin_type == GTINType.GTIN_13: result.company_prefix = cleaned[:7] elif gtin_type == GTINType.GTIN_14: result.company_prefix = cleaned[1:8] # skip indicator digit @@ -403,15 +415,17 @@ def validate_batch(gtins: list[str]) -> dict: ] # --- Duplicate detection --- - cleaned_list = [r.cleaned for r in results if r.cleaned] - counts = Counter(cleaned_list) - duplicates = {k: v for k, v in counts.items() if v > 1} + cleaned_to_rows: dict[str, list[int]] = defaultdict(list) + for r in results: + if r.cleaned: + cleaned_to_rows[r.cleaned].append(r.row_number) + duplicates = {k: len(v) for k, v in cleaned_to_rows.items() if len(v) > 1} for result in results: if result.cleaned in duplicates: other_rows = [ - r.row_number for r in results - if r.cleaned == result.cleaned and r.row_number != result.row_number + rn for rn in cleaned_to_rows[result.cleaned] + if rn != result.row_number ] result.issues.append(Issue( severity=Severity.WARNING, @@ -442,9 +456,11 @@ def validate_batch(gtins: list[str]) -> dict: severity=Severity.WARNING, code="PREFIX_MISMATCH", message=( - f"This GTIN's company prefix ({result.company_prefix}) differs from " - f"the most common prefix in your file ({dominant_prefix}, used by " - f"{dominant_count} of {len(prefixes)} GTINs)." + f"This GTIN's company prefix (~{result.company_prefix}) differs from " + f"the most common prefix in your file (~{dominant_prefix}, used by " + f"{dominant_count} of {len(prefixes)} GTINs). " + "Note: prefixes are detected heuristically using the first 7 digits; " + "actual GS1 company prefix lengths vary from 7 to 10 digits." ), recommendation=( "This could mean: (1) you acquired this product from another company, " @@ -603,13 +619,30 @@ def generate_retailer_checklists( Each check includes a list of failing GTINs (row_number, raw_input) for drill-down in reports. """ + # Precompute the batch-level slices that don't depend on the retailer + # profile — avoids re-iterating `results` once per retailer. + invalid = [r for r in results if not r.is_valid] + dups = [r for r in results if any(i.code == "DUPLICATE" for i in r.issues)] + dup_count = len({r.cleaned for r in dups}) + prefix_failing = [ + r for r in results + if any(i.code == "PREFIX_MISMATCH" for i in r.issues) + ] + has_case = any( + r.gtin_type == GTINType.GTIN_14 and r.indicator_digit in "12345678" + for r in results + ) + + invalid_failing = [(r.row_number, r.raw_input) for r in invalid] + dup_failing = [(r.row_number, r.raw_input) for r in dups] + prefix_failing_rows = [(r.row_number, r.raw_input) for r in prefix_failing] + checklists = {} for retailer_name, profile in RETAILER_PROFILES.items(): checks = [] # Check 1: All GTINs valid - invalid = [r for r in results if not r.is_valid] checks.append({ "check": "All GTINs pass check digit validation", "passed": len(invalid) == 0, @@ -617,12 +650,10 @@ def generate_retailer_checklists( f"{len(invalid)} GTIN(s) have invalid check digits" if invalid else "All check digits valid" ), - "failing_gtins": [(r.row_number, r.raw_input) for r in invalid], + "failing_gtins": invalid_failing, }) # Check 2: No duplicates - dups = [r for r in results if any(i.code == "DUPLICATE" for i in r.issues)] - dup_count = len({r.cleaned for r in dups}) checks.append({ "check": "No duplicate GTINs", "passed": dup_count == 0, @@ -630,16 +661,17 @@ def generate_retailer_checklists( f"{dup_count} duplicate GTIN(s) found" if dup_count else "No duplicates" ), - "failing_gtins": [(r.row_number, r.raw_input) for r in dups], + "failing_gtins": dup_failing, }) - # Check 3: Accepted GTIN types + # Check 3: Accepted GTIN types (profile-dependent) + required_types = profile["required_gtin_types"] wrong_type = [ r for r in results - if r.gtin_type not in profile["required_gtin_types"] + if r.gtin_type not in required_types and r.gtin_type != GTINType.UNKNOWN ] - accepted = ", ".join(t.value for t in profile["required_gtin_types"]) + accepted = ", ".join(t.value for t in required_types) checks.append({ "check": f"GTIN types accepted by {retailer_name}", "passed": len(wrong_type) == 0, @@ -665,10 +697,6 @@ def generate_retailer_checklists( # Check 5: Case GTIN present (if required) if profile["requires_case_gtin"]: - has_case = any( - r.gtin_type == GTINType.GTIN_14 and r.indicator_digit in "12345678" - for r in results - ) checks.append({ "check": "Case-level GTIN-14 present", "passed": has_case, @@ -681,10 +709,6 @@ def generate_retailer_checklists( }) # Check 6: Consistent company prefix - prefix_failing = [ - r for r in results - if any(i.code == "PREFIX_MISMATCH" for i in r.issues) - ] checks.append({ "check": "Consistent GS1 company prefix", "passed": len(prefix_failing) == 0, @@ -693,7 +717,7 @@ def generate_retailer_checklists( if prefix_failing else "All GTINs share a consistent company prefix" ), - "failing_gtins": [(r.row_number, r.raw_input) for r in prefix_failing], + "failing_gtins": prefix_failing_rows, }) passed = sum(1 for c in checks if c["passed"]) @@ -1212,9 +1236,11 @@ def check_data_completeness(df: pd.DataFrame) -> dict: field_analysis: dict[str, dict] = {} for field_name, col in matched_columns.items(): - non_empty = int(df[col].apply( - lambda x: bool(str(x).strip()) if pd.notna(x) else False - ).sum()) + # Vectorized "non-empty after strip" count — replaces a per-row + # Python lambda that scaled poorly on wide product masters. + non_empty = int( + df[col].fillna("").astype(str).str.strip().astype(bool).sum() + ) field_analysis[field_name] = { "column_name": col, diff --git a/pdf_report.py b/pdf_report.py index 6b99645..8fdd6d6 100644 --- a/pdf_report.py +++ b/pdf_report.py @@ -15,10 +15,16 @@ from io import BytesIO from datetime import datetime from collections import defaultdict +from xml.sax.saxutils import escape as _xml_escape + from gtin_core import Severity -# Colors +# ============================================================================= +# Constants +# ============================================================================= + +# Brand palette DARK = colors.HexColor("#1a1a2e") ACCENT = colors.HexColor("#e94560") GRAY = colors.HexColor("#6c757d") @@ -26,170 +32,244 @@ GREEN = colors.HexColor("#28a745") YELLOW = colors.HexColor("#ffc107") RED = colors.HexColor("#dc3545") +INFO_BLUE = colors.HexColor("#17a2b8") WHITE = colors.white - -# Approximate page height available for content (letter = 792pt, minus margins and buffer) -PAGE_CONTENT_HEIGHT = 792 - (0.75 * 72 * 2) - 40 # ~600pt usable +BORDER = colors.HexColor("#dee2e6") +TOTAL_ROW_BG = colors.HexColor("#fff3cd") + +# Page geometry — letter is 792pt tall, with 0.75in margins top+bottom and a +# 40pt safety buffer this leaves ~644pt of usable vertical space we use to +# decide whether a flowable group fits on one page or needs continuation. +PAGE_MARGIN = 0.75 * inch +PAGE_CONTENT_HEIGHT = 792 - (PAGE_MARGIN * 2) - 40 # ~644pt usable + +# Layout heuristics for the item-detail pagination decisions. These are +# rough estimates of rendered flowable height in points; they only need to +# be in the right ballpark for KeepTogether vs explicit continuation. +ITEM_HEADER_PT = 18 +ISSUE_BLOCK_PT = 30 # message + fix line +GROUP_HEADER_PT = 40 # heading + recommendation +MULTI_GROUP_HEADER_PT = 30 +FAILING_ROW_PT = 14 +FAILING_HEADER_PT = 20 + +# Issue-code → human-readable group label, used when grouping items in +# the per-severity sections of the item detail page. +_CRITICAL_LABELS = { + "EMPTY": "Empty or Blank GTINs", + "NON_NUMERIC": "Non-Numeric Characters in GTIN", + "INVALID_LENGTH": "Invalid GTIN Length", + "BAD_CHECK_DIGIT": "Incorrect Check Digit", + "ALL_ZEROS": "Placeholder GTINs (All Zeros)", +} +_WARNING_LABELS = { + "DUPLICATE": "Duplicate GTINs", + "PREFIX_MISMATCH": "Company Prefix Mismatch", + "ORPHAN_CASE_GTIN": "Orphan Case GTINs (no matching unit)", + "INDICATOR_NINE": "Variable Measure Indicator Digit", + "UPC_NOT_GTIN13": "UPC-A Format (GTIN-13 may be required)", + "NO_CASE_GTIN": "Missing Case-Level GTIN-14", +} +_INFO_LABELS = { + "INDICATOR_ZERO": "GTIN-14 with Indicator 0 (base unit in 14-digit format)", + "CASE_LEVEL": "Case/Inner Pack Level GTIN-14", +} + + +# ============================================================================= +# Helpers +# ============================================================================= + +def _escape(value) -> str: + """Escape user-supplied text before embedding in a ReportLab Paragraph. + + ReportLab parses inline XML/HTML-style markup in Paragraph strings, + so any `<`, `>`, or `&` from user input would corrupt rendering or + inject unintended markup. + """ + return _xml_escape("" if value is None else str(value)) def severity_color(severity): + """Map a Severity enum to its brand colour.""" if severity == Severity.CRITICAL: return RED elif severity == Severity.WARNING: return YELLOW - return colors.HexColor("#17a2b8") + return INFO_BLUE -def generate_pdf_report(validation_data: dict, company_name: str = "") -> BytesIO: - """Generate a branded PDF report and return as BytesIO.""" - buffer = BytesIO() - doc = SimpleDocTemplate( - buffer, - pagesize=letter, - topMargin=0.75 * inch, - bottomMargin=0.75 * inch, - leftMargin=0.75 * inch, - rightMargin=0.75 * inch, - ) - - styles = getSampleStyleSheet() - - # Custom styles - title_style = ParagraphStyle( - "ReportTitle", - parent=styles["Title"], - fontSize=22, - textColor=DARK, - spaceAfter=6, - alignment=TA_LEFT, - ) - subtitle_style = ParagraphStyle( - "ReportSubtitle", - parent=styles["Normal"], - fontSize=11, - textColor=GRAY, - spaceAfter=20, - ) - heading_style = ParagraphStyle( - "SectionHeading", - parent=styles["Heading2"], - fontSize=14, - textColor=DARK, - spaceBefore=20, - spaceAfter=10, - borderWidth=0, - ) - body_style = ParagraphStyle( - "BodyText", - parent=styles["Normal"], - fontSize=10, - textColor=DARK, - spaceAfter=6, - leading=14, - ) - small_style = ParagraphStyle( - "SmallText", - parent=styles["Normal"], - fontSize=8, - textColor=GRAY, - spaceAfter=4, - ) - score_style = ParagraphStyle( - "ScoreText", - parent=styles["Normal"], - fontSize=36, - textColor=DARK, - alignment=TA_CENTER, - spaceAfter=4, - leading=44, - ) - grade_style = ParagraphStyle( - "GradeText", - parent=styles["Normal"], - fontSize=16, - textColor=GRAY, - alignment=TA_CENTER, - spaceBefore=16, - spaceAfter=16, - ) - - elements = [] - summary = validation_data["summary"] - score = validation_data["score"] - cost = validation_data["cost_estimate"] - results = validation_data["results"] - - # --- Title page content --- - report_title = "Product Data Validation Report" - if company_name: - report_title = f"Product Data Validation Report" - elements.append(Paragraph(company_name, ParagraphStyle( - "CompanyName", parent=styles["Normal"], - fontSize=12, textColor=ACCENT, spaceAfter=4, - ))) - - elements.append(Paragraph(report_title, title_style)) - elements.append(Paragraph( - f"Generated {datetime.now().strftime('%B %d, %Y at %I:%M %p')}", - subtitle_style, - )) - elements.append(HRFlowable( - width="100%", thickness=1, color=colors.HexColor("#dee2e6"), - spaceAfter=20, - )) - - # --- Readiness Score --- - elements.append(Paragraph("Submission Readiness Score", heading_style)) - - score_color = GREEN if score["score"] >= 75 else (YELLOW if score["score"] >= 50 else RED) - elements.append(Paragraph( - f'{score["score"]}' - f' / 100', - score_style, - )) - elements.append(Spacer(1, 20)) - elements.append(Paragraph(f'Grade: {score["grade"]}', grade_style)) - elements.append(Paragraph(score["interpretation"], body_style)) - elements.append(Spacer(1, 12)) - - # --- Summary table --- - elements.append(Paragraph("Summary", heading_style)) - summary_data = [ - ["Metric", "Value"], - ["Total GTINs Analyzed", str(summary["total_gtins"])], - ["Valid GTINs", str(summary["valid"])], - ["Critical Issues", str(summary["critical_issues"])], - ["Warnings", str(summary["warnings"])], - ["Clean (No Issues)", str(summary["clean"])], - ["Duplicate Groups", str(summary["duplicate_groups"])], - ["Unique Company Prefixes", str(summary["unique_prefixes"])], - ] - summary_table = Table(summary_data, colWidths=[3.5 * inch, 2 * inch]) - summary_table.setStyle(TableStyle([ - ("BACKGROUND", (0, 0), (-1, 0), DARK), - ("TEXTCOLOR", (0, 0), (-1, 0), WHITE), - ("FONTSIZE", (0, 0), (-1, -1), 10), - ("FONTNAME", (0, 0), (-1, 0), "Helvetica-Bold"), - ("ALIGN", (1, 0), (1, -1), "CENTER"), - ("BOTTOMPADDING", (0, 0), (-1, -1), 6), - ("TOPPADDING", (0, 0), (-1, -1), 6), - ("GRID", (0, 0), (-1, -1), 0.5, colors.HexColor("#dee2e6")), - ("ROWBACKGROUNDS", (0, 1), (-1, -1), [WHITE, LIGHT_GRAY]), - ])) - elements.append(summary_table) - elements.append(Spacer(1, 12)) - - # --- Cost of Inaction --- - if cost: - elements.append(PageBreak()) - elements.append(Paragraph("Estimated Cost of Inaction", heading_style)) - elements.append(Paragraph( +def _build_styles() -> dict: + """Build the named ParagraphStyle objects used throughout the report.""" + base = getSampleStyleSheet() + return { + "title": ParagraphStyle( + "ReportTitle", parent=base["Title"], fontSize=22, + textColor=DARK, spaceAfter=6, alignment=TA_LEFT, + ), + "subtitle": ParagraphStyle( + "ReportSubtitle", parent=base["Normal"], fontSize=11, + textColor=GRAY, spaceAfter=20, + ), + "heading": ParagraphStyle( + "SectionHeading", parent=base["Heading2"], fontSize=14, + textColor=DARK, spaceBefore=20, spaceAfter=10, borderWidth=0, + ), + "body": ParagraphStyle( + "BodyText", parent=base["Normal"], fontSize=10, + textColor=DARK, spaceAfter=6, leading=14, + ), + "small": ParagraphStyle( + "SmallText", parent=base["Normal"], fontSize=8, + textColor=GRAY, spaceAfter=4, + ), + "score": ParagraphStyle( + "ScoreText", parent=base["Normal"], fontSize=36, + textColor=DARK, alignment=TA_CENTER, spaceAfter=4, leading=44, + ), + "grade": ParagraphStyle( + "GradeText", parent=base["Normal"], fontSize=16, + textColor=GRAY, alignment=TA_CENTER, + spaceBefore=16, spaceAfter=16, + ), + "company_name": ParagraphStyle( + "CompanyName", parent=base["Normal"], fontSize=12, + textColor=ACCENT, spaceAfter=4, + ), + } + + +# ============================================================================= +# Builder +# ============================================================================= + +class PDFReportBuilder: + """Assemble the validation PDF. + + Each `_render_*` method appends flowables to `self.elements`. The public + `build()` method composes them in order and returns a populated BytesIO. + """ + + def __init__(self, validation_data: dict, company_name: str = ""): + self.data = validation_data + self.company_name = company_name + self.styles = _build_styles() + self.elements: list = [] + self.results = validation_data["results"] + self.summary = validation_data["summary"] + self.score = validation_data["score"] + self.cost = validation_data["cost_estimate"] + self.retailer_checklists = validation_data["retailer_checklists"] + + # -- public entrypoint ------------------------------------------------- + + def build(self) -> BytesIO: + buffer = BytesIO() + doc = SimpleDocTemplate( + buffer, + pagesize=letter, + topMargin=PAGE_MARGIN, + bottomMargin=PAGE_MARGIN, + leftMargin=PAGE_MARGIN, + rightMargin=PAGE_MARGIN, + ) + + self._render_title() + self._render_score() + self._render_summary() + self._render_cost() + self._render_retailer_checklists() + self._render_item_detail() + self._render_footer() + + doc.build(self.elements) + buffer.seek(0) + return buffer + + # -- sections ---------------------------------------------------------- + + def _render_title(self): + if self.company_name: + self.elements.append(Paragraph( + _escape(self.company_name), self.styles["company_name"], + )) + self.elements.append(Paragraph( + "Product Data Validation Report", self.styles["title"], + )) + self.elements.append(Paragraph( + f"Generated {datetime.now().strftime('%B %d, %Y at %I:%M %p')}", + self.styles["subtitle"], + )) + self.elements.append(HRFlowable( + width="100%", thickness=1, color=BORDER, spaceAfter=20, + )) + + def _render_score(self): + score = self.score + self.elements.append(Paragraph( + "Submission Readiness Score", self.styles["heading"], + )) + score_color = ( + GREEN if score["score"] >= 75 + else YELLOW if score["score"] >= 50 + else RED + ) + self.elements.append(Paragraph( + f'{score["score"]}' + f' / 100', + self.styles["score"], + )) + self.elements.append(Spacer(1, 20)) + self.elements.append(Paragraph( + f'Grade: {score["grade"]}', self.styles["grade"], + )) + self.elements.append(Paragraph( + score["interpretation"], self.styles["body"], + )) + self.elements.append(Spacer(1, 12)) + + def _render_summary(self): + s = self.summary + self.elements.append(Paragraph("Summary", self.styles["heading"])) + rows = [ + ["Metric", "Value"], + ["Total GTINs Analyzed", str(s["total_gtins"])], + ["Valid GTINs", str(s["valid"])], + ["Critical Issues", str(s["critical_issues"])], + ["Warnings", str(s["warnings"])], + ["Clean (No Issues)", str(s["clean"])], + ["Duplicate Groups", str(s["duplicate_groups"])], + ["Unique Company Prefixes", str(s["unique_prefixes"])], + ] + table = Table(rows, colWidths=[3.5 * inch, 2 * inch]) + table.setStyle(TableStyle([ + ("BACKGROUND", (0, 0), (-1, 0), DARK), + ("TEXTCOLOR", (0, 0), (-1, 0), WHITE), + ("FONTSIZE", (0, 0), (-1, -1), 10), + ("FONTNAME", (0, 0), (-1, 0), "Helvetica-Bold"), + ("ALIGN", (1, 0), (1, -1), "CENTER"), + ("BOTTOMPADDING", (0, 0), (-1, -1), 6), + ("TOPPADDING", (0, 0), (-1, -1), 6), + ("GRID", (0, 0), (-1, -1), 0.5, BORDER), + ("ROWBACKGROUNDS", (0, 1), (-1, -1), [WHITE, LIGHT_GRAY]), + ])) + self.elements.append(table) + self.elements.append(Spacer(1, 12)) + + def _render_cost(self): + cost = self.cost + if not cost: + return + self.elements.append(PageBreak()) + self.elements.append(Paragraph( + "Estimated Cost of Inaction", self.styles["heading"], + )) + self.elements.append(Paragraph( "These estimates are based on industry averages for specialty food brands " "at similar scale. Actual costs vary by retailer mix and volume.", - small_style, + self.styles["small"], )) - - cost_data = [ + rows = [ ["Cost Category", "Estimated Annual Range"], [ "Chargebacks from invalid GTINs", @@ -208,389 +288,409 @@ def generate_pdf_report(validation_data: dict, company_name: str = "") -> BytesI f"${cost['annual_estimate_low']:,} – ${cost['annual_estimate_high']:,}", ], ] - cost_table = Table(cost_data, colWidths=[3.5 * inch, 2.5 * inch]) - cost_table.setStyle(TableStyle([ + table = Table(rows, colWidths=[3.5 * inch, 2.5 * inch]) + table.setStyle(TableStyle([ ("BACKGROUND", (0, 0), (-1, 0), DARK), ("TEXTCOLOR", (0, 0), (-1, 0), WHITE), ("FONTSIZE", (0, 0), (-1, -1), 10), ("FONTNAME", (0, 0), (-1, 0), "Helvetica-Bold"), ("FONTNAME", (0, -1), (-1, -1), "Helvetica-Bold"), - ("BACKGROUND", (0, -1), (-1, -1), colors.HexColor("#fff3cd")), + ("BACKGROUND", (0, -1), (-1, -1), TOTAL_ROW_BG), ("ALIGN", (1, 0), (1, -1), "RIGHT"), ("BOTTOMPADDING", (0, 0), (-1, -1), 6), ("TOPPADDING", (0, 0), (-1, -1), 6), - ("GRID", (0, 0), (-1, -1), 0.5, colors.HexColor("#dee2e6")), + ("GRID", (0, 0), (-1, -1), 0.5, BORDER), ("ROWBACKGROUNDS", (0, 1), (-1, -2), [WHITE, LIGHT_GRAY]), ])) - elements.append(cost_table) - + self.elements.append(table) if cost.get("growth_note"): - elements.append(Spacer(1, 6)) - elements.append(Paragraph(f"{cost['growth_note']}", small_style)) - - # --- Retailer Checklists --- - retailer_checklists = validation_data["retailer_checklists"] - total_gtins = validation_data["summary"]["total_gtins"] - for idx, (retailer_name, checklist) in enumerate(retailer_checklists.items()): - elements.append(PageBreak()) - if idx == 0: - elements.append(Paragraph("Retailer Readiness Checklists", heading_style)) - elements.append(Spacer(1, 12)) - - if checklist["ready"]: - status = "ALL CHECKS PASSED" - else: - status = ( - f"{checklist['passed']} of {checklist['total']} GTIN validation checks passed " - f"(across all {total_gtins} GTINs submitted)" - ) - status_color = GREEN if checklist["ready"] else RED + self.elements.append(Spacer(1, 6)) + self.elements.append(Paragraph( + f"{_escape(cost['growth_note'])}", self.styles["small"], + )) - elements.append(Paragraph( - f' ' - f'{retailer_name}', - ParagraphStyle("RetailerName", parent=body_style, fontSize=14, spaceBefore=12), - )) - elements.append(Paragraph( - checklist["profile"]["description"], - small_style, - )) - elements.append(Paragraph( - f'{status}', - ParagraphStyle("RetailerStatus", parent=body_style, fontSize=10, - textColor=status_color, spaceAfter=8), - )) + def _render_retailer_checklists(self): + total_gtins = self.summary["total_gtins"] + for idx, (retailer_name, checklist) in enumerate(self.retailer_checklists.items()): + self.elements.append(PageBreak()) + if idx == 0: + self.elements.append(Paragraph( + "Retailer Readiness Checklists", self.styles["heading"], + )) + self.elements.append(Spacer(1, 12)) + + if checklist["ready"]: + status = "ALL CHECKS PASSED" + else: + status = ( + f"{checklist['passed']} of {checklist['total']} GTIN validation checks passed " + f"(across all {total_gtins} GTINs submitted)" + ) + status_color = GREEN if checklist["ready"] else RED - # List each check - for check in checklist["checks"]: - icon = "✓" if check["passed"] else "✗" - icon_color = GREEN if check["passed"] else RED - elements.append(Paragraph( - f'{icon} ' - f'{check["check"]} — {check["detail"]}', - ParagraphStyle("CheckItem", parent=body_style, fontSize=9, leftIndent=20), + self.elements.append(Paragraph( + f' ' + f'{retailer_name}', + ParagraphStyle("RetailerName", parent=self.styles["body"], + fontSize=14, spaceBefore=12), )) - - # Group failing GTINs by check (issue type) - failed_checks = [c for c in checklist["checks"] if not c["passed"] and c.get("failing_gtins")] - if failed_checks: - elements.append(Spacer(1, 12)) - elements.append(Paragraph( - 'Failing GTINs by Issue', - ParagraphStyle("FailingHeader", parent=body_style, fontSize=11, spaceBefore=8), + self.elements.append(Paragraph( + checklist["profile"]["description"], self.styles["small"], + )) + self.elements.append(Paragraph( + f'{status}', + ParagraphStyle("RetailerStatus", parent=self.styles["body"], + fontSize=10, textColor=status_color, spaceAfter=8), )) - for check in failed_checks: - failing = check["failing_gtins"] - if not failing: - continue - - # Try to keep the whole group together - block = [] - block.append(Paragraph( - f' {check["check"]} — ' - f'{len(failing)} GTIN(s)', - ParagraphStyle("FailGroup", parent=body_style, fontSize=10, - spaceBefore=10, leftIndent=20), + for check in checklist["checks"]: + icon = "✓" if check["passed"] else "✗" + icon_color = GREEN if check["passed"] else RED + self.elements.append(Paragraph( + f'{icon} ' + f'{check["check"]} — {check["detail"]}', + ParagraphStyle("CheckItem", parent=self.styles["body"], + fontSize=9, leftIndent=20), )) - for row_num, raw_input in failing: - block.append(Paragraph( - f'Row {row_num}: {raw_input}', - ParagraphStyle("FailItem", parent=small_style, fontSize=8, - leftIndent=40), - )) - - # Estimate height: header ~20pt + each row ~14pt - est_height = 20 + len(failing) * 14 - if est_height <= PAGE_CONTENT_HEIGHT: - elements.append(KeepTogether(block)) - else: - # Too tall — chunk with continued headers - elements.append(block[0]) # header - running = 20 - for j, item_para in enumerate(block[1:]): - if running + 14 > PAGE_CONTENT_HEIGHT and j > 0: - elements.append(PageBreak()) - elements.append(Paragraph( - f' ' - f'{check["check"]} — continued', - ParagraphStyle("FailGroupCont", parent=body_style, - fontSize=10, spaceBefore=10, leftIndent=20), - )) - running = 20 - elements.append(item_para) - running += 14 - - # --- Item Detail --- - elements.append(PageBreak()) - elements.append(Paragraph("Item-Level Detail", heading_style)) - elements.append(Paragraph( - "Each GTIN analyzed, with issues and recommendations.", - small_style, - )) - - # Sort: critical first, then warnings, then clean - sorted_results = sorted( - results, - key=lambda r: ( - 0 if r.has_critical else (1 if r.has_warning else 2), - r.row_number, - ), - ) - # Split results by severity - critical_items = [r for r in sorted_results if r.has_critical] - warning_items = [r for r in sorted_results if r.has_warning and not r.has_critical] - info_items = [r for r in sorted_results if r.issues and not r.has_critical and not r.has_warning] + self._render_failing_gtins_for_retailer(checklist) - def render_item_flowables(r, label_color, body_style, small_style): - """Return a list of flowables for one row (NOT wrapped in KeepTogether).""" - block = [] - block.append(Paragraph( - f' ' - f'Row {r.row_number}: {r.raw_input} ' - f'({r.gtin_type.value if r.gtin_type.value != "Unknown" else "Unknown format"})', - ParagraphStyle("ItemHeader", parent=body_style, fontSize=10, spaceBefore=10), + def _render_failing_gtins_for_retailer(self, checklist): + failed_checks = [ + c for c in checklist["checks"] + if not c["passed"] and c.get("failing_gtins") + ] + if not failed_checks: + return + + self.elements.append(Spacer(1, 12)) + self.elements.append(Paragraph( + 'Failing GTINs by Issue', + ParagraphStyle("FailingHeader", parent=self.styles["body"], + fontSize=11, spaceBefore=8), )) - for issue in r.issues: - block.append(Paragraph( - f'[{issue.severity.value}] {issue.message}', - ParagraphStyle("IssueMsg", parent=body_style, fontSize=9, leftIndent=20), - )) - block.append(Paragraph( - f'Fix: {issue.recommendation}', - ParagraphStyle("IssueFix", parent=small_style, leftIndent=20), - )) - return block - def render_item_block(r, label_color, body_style, small_style): - """Return a KeepTogether block for one row.""" - return KeepTogether(render_item_flowables(r, label_color, body_style, small_style)) - - def estimate_item_height(r): - """Rough estimate of how tall one item block will be in points.""" - # Header line ~18pt + each issue ~30pt (message + fix) - return 18 + len(r.issues) * 30 - - def render_group_with_continuation(group_label, recommendation_text, items, - label_color, body_style, small_style, elements): - """ - Render a group of items. Try to keep header + all items together. - If too tall, chunk into pages with 'continued' headers. - """ - # Build the header flowables - def make_header(continued=False): - suffix = " — continued" if continued else "" - header_parts = [] - header_parts.append(Paragraph( - f'{group_label}{suffix} — {len(items)} item(s)', - ParagraphStyle("GroupHeader", parent=body_style, fontSize=11, - spaceBefore=16, spaceAfter=4, textColor=DARK), - )) - if recommendation_text and not continued: - header_parts.append(Paragraph( - f'{recommendation_text}', - ParagraphStyle("GroupRec", parent=small_style, leftIndent=20, spaceAfter=8), + for check in failed_checks: + failing = check["failing_gtins"] + if not failing: + continue + + block = [Paragraph( + f' {check["check"]} — ' + f'{len(failing)} GTIN(s)', + ParagraphStyle("FailGroup", parent=self.styles["body"], + fontSize=10, spaceBefore=10, leftIndent=20), + )] + for row_num, raw_input in failing: + block.append(Paragraph( + f'Row {row_num}: {_escape(raw_input)}', + ParagraphStyle("FailItem", parent=self.styles["small"], + fontSize=8, leftIndent=40), )) - return header_parts - # Calculate total height - total_height = 40 # header + recommendation - item_heights = [] - for r in items: - h = estimate_item_height(r) - item_heights.append(h) - total_height += h + est_height = FAILING_HEADER_PT + len(failing) * FAILING_ROW_PT + if est_height <= PAGE_CONTENT_HEIGHT: + self.elements.append(KeepTogether(block)) + else: + # Too tall — chunk with continued headers + self.elements.append(block[0]) + running = FAILING_HEADER_PT + for j, item_para in enumerate(block[1:]): + if running + FAILING_ROW_PT > PAGE_CONTENT_HEIGHT and j > 0: + self.elements.append(PageBreak()) + self.elements.append(Paragraph( + f' ' + f'{check["check"]} — continued', + ParagraphStyle("FailGroupCont", parent=self.styles["body"], + fontSize=10, spaceBefore=10, leftIndent=20), + )) + running = FAILING_HEADER_PT + self.elements.append(item_para) + running += FAILING_ROW_PT + + def _render_item_detail(self): + self.elements.append(PageBreak()) + self.elements.append(Paragraph( + "Item-Level Detail", self.styles["heading"], + )) + self.elements.append(Paragraph( + "Each GTIN analyzed, with issues and recommendations.", + self.styles["small"], + )) - # If everything fits on one page, wrap it all in KeepTogether - if total_height <= PAGE_CONTENT_HEIGHT: - group_block = make_header(continued=False) - for r in items: - group_block.extend(render_item_flowables(r, label_color, body_style, small_style)) - elements.append(KeepTogether(group_block)) - else: - # Too tall for one page — chunk with continued headers - elements.extend(make_header(continued=False)) - - running_height = 40 # header already placed - for i, r in enumerate(items): - h = item_heights[i] - if running_height + h > PAGE_CONTENT_HEIGHT and i > 0: - # Start new page with continued header - elements.append(PageBreak()) - elements.extend(make_header(continued=True)) - running_height = 40 - elements.append(render_item_block(r, label_color, body_style, small_style)) - running_height += h - - def render_multi_issue_group(group_label, items, label_color, body_style, small_style, elements): - """Render a multi-issue group with continuation support.""" - def make_header(continued=False): - suffix = " — continued" if continued else "" - return [Paragraph( - f'{group_label}{suffix} — {len(items)} item(s)', - ParagraphStyle("MultiGroupHeader", parent=body_style, fontSize=11, - spaceBefore=16, spaceAfter=8, textColor=DARK), - )] + sorted_results = sorted( + self.results, + key=lambda r: ( + 0 if r.has_critical else (1 if r.has_warning else 2), + r.row_number, + ), + ) + + critical_items = [r for r in sorted_results if r.has_critical] + warning_items = [ + r for r in sorted_results + if r.has_warning and not r.has_critical + ] + info_items = [ + r for r in sorted_results + if r.issues and not r.has_critical and not r.has_warning + ] - total_height = 30 - item_heights = [estimate_item_height(r) for r in items] - total_height += sum(item_heights) + self._render_critical_section(critical_items) + self._render_warning_section(warning_items) + self._render_info_section(info_items) + self._render_clean_summary() - if total_height <= PAGE_CONTENT_HEIGHT: - group_block = make_header(continued=False) - for r in items: - group_block.extend(render_item_flowables(r, label_color, body_style, small_style)) - elements.append(KeepTogether(group_block)) - else: - elements.extend(make_header(continued=False)) - running_height = 30 - for i, r in enumerate(items): - h = item_heights[i] - if running_height + h > PAGE_CONTENT_HEIGHT and i > 0: - elements.append(PageBreak()) - elements.extend(make_header(continued=True)) - running_height = 30 - elements.append(render_item_block(r, label_color, body_style, small_style)) - running_height += h - - # --- Critical Issues — grouped by issue type --- - if critical_items: - elements.append(Paragraph( + def _render_critical_section(self, critical_items): + if not critical_items: + return + self.elements.append(Paragraph( f' ' f'Critical Issues — These GTINs will be rejected', - ParagraphStyle("SeverityHeader", parent=heading_style, fontSize=14), + ParagraphStyle("SeverityHeader", parent=self.styles["heading"], fontSize=14), )) - elements.append(Spacer(1, 8)) + self.elements.append(Spacer(1, 8)) - single_critical = [r for r in critical_items if len([i for i in r.issues if i.severity == Severity.CRITICAL]) == 1] - multi_critical = [r for r in critical_items if len([i for i in r.issues if i.severity == Severity.CRITICAL]) > 1] + single_critical = [ + r for r in critical_items + if sum(1 for i in r.issues if i.severity == Severity.CRITICAL) == 1 + ] + multi_critical = [ + r for r in critical_items + if sum(1 for i in r.issues if i.severity == Severity.CRITICAL) > 1 + ] if single_critical: - crit_groups = defaultdict(list) + groups: dict[str, list] = defaultdict(list) for r in single_critical: - crit_issue = next(i for i in r.issues if i.severity == Severity.CRITICAL) - crit_groups[crit_issue.code].append(r) - - crit_code_labels = { - "EMPTY": "Empty or Blank GTINs", - "NON_NUMERIC": "Non-Numeric Characters in GTIN", - "INVALID_LENGTH": "Invalid GTIN Length", - "BAD_CHECK_DIGIT": "Incorrect Check Digit", - "ALL_ZEROS": "Placeholder GTINs (All Zeros)", - } - - for code, items in crit_groups.items(): - group_label = crit_code_labels.get(code, code) + crit_issue = next( + i for i in r.issues if i.severity == Severity.CRITICAL + ) + groups[crit_issue.code].append(r) + for code, items in groups.items(): + label = _CRITICAL_LABELS.get(code, code) sample_issue = next(i for i in items[0].issues if i.code == code) - render_group_with_continuation( - group_label, sample_issue.recommendation, items, - RED, body_style, small_style, elements, + self._render_group_with_continuation( + label, sample_issue.recommendation, items, RED, ) if multi_critical: - multi_critical.sort(key=lambda r: len([i for i in r.issues if i.severity == Severity.CRITICAL]), reverse=True) - render_multi_issue_group( - "Items with Multiple Critical Issues", multi_critical, - RED, body_style, small_style, elements, + multi_critical.sort( + key=lambda r: sum(1 for i in r.issues if i.severity == Severity.CRITICAL), + reverse=True, + ) + self._render_multi_issue_group( + "Items with Multiple Critical Issues", multi_critical, RED, ) - # --- Warnings — grouped by issue type --- - if warning_items: - elements.append(PageBreak()) - elements.append(Paragraph( + def _render_warning_section(self, warning_items): + if not warning_items: + return + self.elements.append(PageBreak()) + self.elements.append(Paragraph( f' ' f'Warnings — These GTINs may cause problems', - ParagraphStyle("SeverityHeader", parent=heading_style, fontSize=14), + ParagraphStyle("SeverityHeader", parent=self.styles["heading"], fontSize=14), )) - elements.append(Spacer(1, 8)) - - single_issue = [r for r in warning_items if len([i for i in r.issues if i.severity == Severity.WARNING]) == 1] - multi_issue = [r for r in warning_items if len([i for i in r.issues if i.severity == Severity.WARNING]) > 1] - - if single_issue: - issue_groups = defaultdict(list) - for r in single_issue: - warning_issue = next(i for i in r.issues if i.severity == Severity.WARNING) - issue_groups[warning_issue.code].append(r) - - code_labels = { - "DUPLICATE": "Duplicate GTINs", - "PREFIX_MISMATCH": "Company Prefix Mismatch", - "ORPHAN_CASE_GTIN": "Orphan Case GTINs (no matching unit)", - "INDICATOR_NINE": "Variable Measure Indicator Digit", - "UPC_NOT_GTIN13": "UPC-A Format (GTIN-13 may be required)", - "NO_CASE_GTIN": "Missing Case-Level GTIN-14", - } - - for code, items in issue_groups.items(): - group_label = code_labels.get(code, code) + self.elements.append(Spacer(1, 8)) + + single = [ + r for r in warning_items + if sum(1 for i in r.issues if i.severity == Severity.WARNING) == 1 + ] + multi = [ + r for r in warning_items + if sum(1 for i in r.issues if i.severity == Severity.WARNING) > 1 + ] + + if single: + groups: dict[str, list] = defaultdict(list) + for r in single: + warning_issue = next( + i for i in r.issues if i.severity == Severity.WARNING + ) + groups[warning_issue.code].append(r) + for code, items in groups.items(): + label = _WARNING_LABELS.get(code, code) sample_issue = next(i for i in items[0].issues if i.code == code) - render_group_with_continuation( - group_label, sample_issue.recommendation, items, - YELLOW, body_style, small_style, elements, + self._render_group_with_continuation( + label, sample_issue.recommendation, items, YELLOW, ) - if multi_issue: - multi_issue.sort(key=lambda r: len([i for i in r.issues if i.severity == Severity.WARNING]), reverse=True) - render_multi_issue_group( - "Items with Multiple Warnings", multi_issue, - YELLOW, body_style, small_style, elements, + if multi: + multi.sort( + key=lambda r: sum(1 for i in r.issues if i.severity == Severity.WARNING), + reverse=True, + ) + self._render_multi_issue_group( + "Items with Multiple Warnings", multi, YELLOW, ) - # --- Info --- - if info_items: - elements.append(PageBreak()) - elements.append(Paragraph( - f' ' + def _render_info_section(self, info_items): + if not info_items: + return + self.elements.append(PageBreak()) + self.elements.append(Paragraph( + f' ' f'Info — Best practice notes', - ParagraphStyle("SeverityHeader", parent=heading_style, fontSize=14), + ParagraphStyle("SeverityHeader", parent=self.styles["heading"], fontSize=14), )) - elements.append(Spacer(1, 8)) + self.elements.append(Spacer(1, 8)) - # Group info items by code too - info_groups = defaultdict(list) + groups: dict[str, list] = defaultdict(list) for r in info_items: - # Use first info issue code for grouping - info_issue = next((i for i in r.issues if i.severity == Severity.INFO), r.issues[0]) - info_groups[info_issue.code].append(r) - - info_code_labels = { - "INDICATOR_ZERO": "GTIN-14 with Indicator 0 (base unit in 14-digit format)", - "CASE_LEVEL": "Case/Inner Pack Level GTIN-14", - } - - for code, items in info_groups.items(): - group_label = info_code_labels.get(code, code) - sample_issue = next((i for i in items[0].issues if i.code == code), items[0].issues[0]) - render_group_with_continuation( - group_label, sample_issue.recommendation, items, - colors.HexColor("#17a2b8"), body_style, small_style, elements, + info_issue = next( + (i for i in r.issues if i.severity == Severity.INFO), + r.issues[0], ) + groups[info_issue.code].append(r) - # --- Clean items summary --- - clean_items = [r for r in results if not r.issues] - if clean_items: - elements.append(Spacer(1, 12)) - elements.append(Paragraph( + for code, items in groups.items(): + label = _INFO_LABELS.get(code, code) + sample_issue = next( + (i for i in items[0].issues if i.code == code), + items[0].issues[0], + ) + self._render_group_with_continuation( + label, sample_issue.recommendation, items, INFO_BLUE, + ) + + def _render_clean_summary(self): + clean_items = [r for r in self.results if not r.issues] + if not clean_items: + return + self.elements.append(Spacer(1, 12)) + self.elements.append(Paragraph( f"{len(clean_items)} GTIN(s) passed all checks with no issues.", - ParagraphStyle("CleanSummary", parent=body_style, textColor=GREEN), + ParagraphStyle("CleanSummary", parent=self.styles["body"], textColor=GREEN), + )) + + def _render_footer(self): + self.elements.append(Spacer(1, 30)) + self.elements.append(HRFlowable( + width="100%", thickness=0.5, color=BORDER, spaceAfter=10, )) + self.elements.append(Paragraph( + "This report was generated by the GTIN Product Data Validator. " + "Estimates are directional based on industry averages and should be " + "validated against your specific retailer relationships and volume. " + "For a comprehensive Product Data Health Audit, contact the author.", + ParagraphStyle("Footer", parent=self.styles["small"], alignment=TA_CENTER), + )) + + # -- per-item rendering helpers --------------------------------------- + + def _render_item_flowables(self, r, label_color): + """Return a list of flowables for one row (NOT wrapped in KeepTogether).""" + block = [Paragraph( + f' ' + f'Row {r.row_number}: {_escape(r.raw_input)} ' + f'({_escape(r.gtin_type.value) if r.gtin_type.value != "Unknown" else "Unknown format"})', + ParagraphStyle("ItemHeader", parent=self.styles["body"], + fontSize=10, spaceBefore=10), + )] + for issue in r.issues: + block.append(Paragraph( + f'[{issue.severity.value}] {_escape(issue.message)}', + ParagraphStyle("IssueMsg", parent=self.styles["body"], + fontSize=9, leftIndent=20), + )) + block.append(Paragraph( + f'Fix: {_escape(issue.recommendation)}', + ParagraphStyle("IssueFix", parent=self.styles["small"], leftIndent=20), + )) + return block + + def _render_item_block(self, r, label_color): + return KeepTogether(self._render_item_flowables(r, label_color)) + + @staticmethod + def _estimate_item_height(r): + return ITEM_HEADER_PT + len(r.issues) * ISSUE_BLOCK_PT + + def _render_group_with_continuation( + self, group_label, recommendation_text, items, label_color, + ): + """Render a group of items. Try KeepTogether; fall back to chunked.""" + def make_header(continued: bool): + suffix = " — continued" if continued else "" + parts = [Paragraph( + f'{group_label}{suffix} — {len(items)} item(s)', + ParagraphStyle("GroupHeader", parent=self.styles["body"], + fontSize=11, spaceBefore=16, spaceAfter=4, + textColor=DARK), + )] + if recommendation_text and not continued: + parts.append(Paragraph( + f'{_escape(recommendation_text)}', + ParagraphStyle("GroupRec", parent=self.styles["small"], + leftIndent=20, spaceAfter=8), + )) + return parts + + item_heights = [self._estimate_item_height(r) for r in items] + total_height = GROUP_HEADER_PT + sum(item_heights) + + if total_height <= PAGE_CONTENT_HEIGHT: + block = make_header(continued=False) + for r in items: + block.extend(self._render_item_flowables(r, label_color)) + self.elements.append(KeepTogether(block)) + return + + # Too tall for one page — chunk with continued headers + self.elements.extend(make_header(continued=False)) + running = GROUP_HEADER_PT + for i, r in enumerate(items): + h = item_heights[i] + if running + h > PAGE_CONTENT_HEIGHT and i > 0: + self.elements.append(PageBreak()) + self.elements.extend(make_header(continued=True)) + running = GROUP_HEADER_PT + self.elements.append(self._render_item_block(r, label_color)) + running += h + + def _render_multi_issue_group(self, group_label, items, label_color): + """Render an item group whose entries each have multiple issues.""" + def make_header(continued: bool): + suffix = " — continued" if continued else "" + return [Paragraph( + f'{group_label}{suffix} — {len(items)} item(s)', + ParagraphStyle("MultiGroupHeader", parent=self.styles["body"], + fontSize=11, spaceBefore=16, spaceAfter=8, + textColor=DARK), + )] + + item_heights = [self._estimate_item_height(r) for r in items] + total_height = MULTI_GROUP_HEADER_PT + sum(item_heights) + + if total_height <= PAGE_CONTENT_HEIGHT: + block = make_header(continued=False) + for r in items: + block.extend(self._render_item_flowables(r, label_color)) + self.elements.append(KeepTogether(block)) + return + + self.elements.extend(make_header(continued=False)) + running = MULTI_GROUP_HEADER_PT + for i, r in enumerate(items): + h = item_heights[i] + if running + h > PAGE_CONTENT_HEIGHT and i > 0: + self.elements.append(PageBreak()) + self.elements.extend(make_header(continued=True)) + running = MULTI_GROUP_HEADER_PT + self.elements.append(self._render_item_block(r, label_color)) + running += h + + +# ============================================================================= +# Public API +# ============================================================================= + +def generate_pdf_report(validation_data: dict, company_name: str = "") -> BytesIO: + """Generate a branded PDF report and return it as BytesIO. - # --- Footer --- - elements.append(Spacer(1, 30)) - elements.append(HRFlowable( - width="100%", thickness=0.5, color=colors.HexColor("#dee2e6"), - spaceAfter=10, - )) - elements.append(Paragraph( - "This report was generated by the GTIN Product Data Validator. " - "Estimates are directional based on industry averages and should be " - "validated against your specific retailer relationships and volume. " - "For a comprehensive Product Data Health Audit, contact the author.", - ParagraphStyle("Footer", parent=small_style, alignment=TA_CENTER), - )) - - doc.build(elements) - buffer.seek(0) - return buffer + This is the stable entry point — the construction is delegated to + PDFReportBuilder, which is internal. + """ + return PDFReportBuilder(validation_data, company_name).build() diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 0000000..9ff54d0 --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,2 @@ +-r requirements.txt +pytest>=7.0,<9.0 diff --git a/requirements.txt b/requirements.txt index f2bb44d..189c907 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,3 +1,3 @@ -streamlit>=1.28.0 -pandas>=2.0.0 -reportlab>=4.0.0 +streamlit>=1.54.0,<2.0 +pandas>=2.0.0,<3.0 +reportlab>=4.0.0,<5.0 diff --git a/tests.py b/tests.py index ceb0da3..be6da3d 100644 --- a/tests.py +++ b/tests.py @@ -4,6 +4,7 @@ Run with: python -m pytest tests.py -v """ +import pandas as pd import pytest from gtin_core import ( calculate_check_digit, @@ -11,13 +12,20 @@ validate_single_gtin, validate_batch, analyze_hierarchy, + calculate_readiness_score, + check_data_completeness, + estimate_cost_of_inaction, generate_before_after, generate_executive_summary, generate_fix_roadmap, generate_gtin14_suggestions, + generate_retailer_checklists, GTINType, + RETAILER_PROFILES, Severity, ) +from csv_report import generate_csv_report +from pdf_report import generate_pdf_report # ========================================================================= @@ -127,8 +135,10 @@ def test_upc_gtin13_warning(self): assert any(i.code == "UPC_NOT_GTIN13" for i in result.issues) def test_company_prefix_extracted_gtin12(self): + # GTIN-12 is normalized onto the GTIN-13 frame (leading zero) before + # the prefix slice is taken, so it matches a paired GTIN-14. result = validate_single_gtin("614141000012", row_number=1) - assert result.company_prefix == "6141410" + assert result.company_prefix == "0614141" def test_company_prefix_extracted_gtin14(self): result = validate_single_gtin("10614141000019", row_number=1) @@ -156,7 +166,7 @@ def test_prefix_mismatch_detection(self): i.code == "PREFIX_MISMATCH" for i in r.issues )] assert len(mismatched) == 1 - assert mismatched[0].company_prefix == "7321410" + assert mismatched[0].company_prefix == "0732141" def test_summary_counts(self): data = validate_batch([ @@ -298,3 +308,288 @@ def test_no_suggestion_for_invalid_gtin(self): data["results"], data["hierarchy"] ) assert len(suggestions) == 0 + + +# ========================================================================= +# Single-GTIN edge cases not covered above +# ========================================================================= + +class TestSingleValidationEdgeCases: + def test_valid_gtin8(self): + # 96385074 is a canonical valid GTIN-8 (check digit 4) + result = validate_single_gtin("96385074", row_number=1) + assert result.gtin_type == GTINType.GTIN_8 + assert result.is_valid + assert not result.has_critical + assert result.company_prefix == "9638" + + def test_gtin14_indicator_zero(self): + # Indicator 0 => base unit in 14-digit form. Check digit: 1061414100002 -> 8 + # We construct a valid GTIN-14 with indicator 0: + from gtin_core import calculate_check_digit + payload = "0061414100001" + check = calculate_check_digit(payload) + gtin14 = payload + str(check) + result = validate_single_gtin(gtin14, row_number=1) + assert result.indicator_digit == "0" + assert any(i.code == "INDICATOR_ZERO" for i in result.issues) + # INDICATOR_ZERO is informational only + assert not result.has_critical + + def test_none_input_handled_gracefully(self): + result = validate_single_gtin(None, row_number=1) + assert not result.is_valid + assert result.issues[0].code == "EMPTY" + + def test_nan_input_handled_gracefully(self): + result = validate_single_gtin(float("nan"), row_number=1) + assert not result.is_valid + assert result.issues[0].code == "EMPTY" + + def test_numeric_input_handled_gracefully(self): + # int / float values are coerced — they should validate normally, + # not crash on .strip() + result = validate_single_gtin(614141000012, row_number=1) + assert result.gtin_type == GTINType.GTIN_12 + assert not result.has_critical + + def test_duplicate_lists_all_sibling_rows(self): + data = validate_batch([ + "614141000012", + "614141000012", + "614141000012", + ]) + # Each duplicate should list the other two row numbers in its message + for r in data["results"]: + dup_issues = [i for i in r.issues if i.code == "DUPLICATE"] + assert len(dup_issues) == 1 + sibling_rows = [ + str(n) for n in (1, 2, 3) if n != r.row_number + ] + for sr in sibling_rows: + assert sr in dup_issues[0].message + + +# ========================================================================= +# Readiness scoring (grade thresholds + bonuses/penalties) +# ========================================================================= + +class TestReadinessScore: + def test_empty_returns_zero(self): + score = calculate_readiness_score([], {"has_hierarchy": False, "hierarchy_complete": False}) + assert score["score"] == 0 + assert score["grade"] == "N/A" + + def test_grade_landscape_with_warnings(self): + # GTIN-12 inputs all carry a UPC_NOT_GTIN13 warning. With a complete + # hierarchy bonus the score lands in the C/B band — and crucially + # never in F since there are no critical issues. + data = validate_batch([ + "614141000012", "614141000029", "614141000036", + "10614141000019", "10614141000026", "10614141000033", + ]) + assert data["score"]["grade"] in {"A", "B", "C"} + assert data["summary"]["critical_issues"] == 0 + + def test_grade_f_on_all_critical(self): + data = validate_batch([ + "61414100010", # invalid length + "6141410003A5", # non-numeric + "000000000000", # all zeros + ]) + assert data["score"]["grade"] == "F" + assert data["score"]["score"] < 40 + + def test_hierarchy_bonus_applied(self): + no_hierarchy = validate_batch(["614141000012"]) + with_hierarchy = validate_batch(["614141000012", "10614141000019"]) + # With a matched unit/case pair the score should be at least as + # high as without (and typically higher because of the bonus) + assert with_hierarchy["score"]["score"] >= no_hierarchy["score"]["score"] + + +# ========================================================================= +# Cost-of-inaction +# ========================================================================= + +class TestCostEstimate: + def test_empty_returns_empty_dict(self): + assert estimate_cost_of_inaction([]) == {} + + def test_low_sku_count_growth_note(self): + # 1 critical, < 20 total: should use the "compound" growth note + data = validate_batch(["61414100010"]) + cost = data["cost_estimate"] + assert "compound" in cost["growth_note"].lower() + + def test_high_sku_count_growth_note(self): + # 20+ inputs: should use the explicit "2x" growth note + gtins = ["614141000012"] * 20 + data = validate_batch(gtins) + cost = data["cost_estimate"] + assert "2x" in cost["growth_note"] + + def test_rework_cost_matches_rework_hours(self): + data = validate_batch(["61414100010", "614141000012"]) + cost = data["cost_estimate"] + assert cost["rework_cost"] == cost["rework_hours"] * 50 + + def test_low_le_high_across_all_ranges(self): + data = validate_batch(["61414100010", "000000000000"]) + cost = data["cost_estimate"] + assert cost["chargeback_range"][0] <= cost["chargeback_range"][1] + assert cost["delayed_launch_range"][0] <= cost["delayed_launch_range"][1] + assert cost["annual_estimate_low"] <= cost["annual_estimate_high"] + + +# ========================================================================= +# Retailer checklists +# ========================================================================= + +class TestRetailerChecklists: + def test_all_profiles_included(self): + data = validate_batch(["614141000012"]) + checklists = data["retailer_checklists"] + for retailer in RETAILER_PROFILES.keys(): + assert retailer in checklists + assert "checks" in checklists[retailer] + assert "ready" in checklists[retailer] + + def test_check_digit_failure_reflected_per_retailer(self): + data = validate_batch(["614141000356"]) # bad check digit + for retailer, cl in data["retailer_checklists"].items(): + cd_check = next(c for c in cl["checks"] if "check digit" in c["check"].lower()) + assert cd_check["passed"] is False + assert cd_check["failing_gtins"] + + def test_clean_data_with_hierarchy_passes_walmart(self): + data = validate_batch([ + "614141000012", # unit + "10614141000019", # matching case + ]) + walmart = data["retailer_checklists"]["Walmart"] + # All check-digit / duplicate / case-present / hierarchy / prefix + # checks pass; UPC types accepted by Walmart + assert walmart["passed"] == walmart["total"] + + def test_hierarchy_check_only_for_retailers_that_require_it(self): + data = validate_batch(["614141000012"]) + for retailer, cl in data["retailer_checklists"].items(): + has_hierarchy_check = any( + "hierarchy" in c["check"].lower() for c in cl["checks"] + ) + profile_requires = cl["profile"]["requires_hierarchy"] + assert has_hierarchy_check == profile_requires + + +# ========================================================================= +# Data completeness +# ========================================================================= + +class TestDataCompleteness: + def test_empty_dataframe(self): + df = pd.DataFrame({"GTIN": []}) + result = check_data_completeness(df) + assert result["field_analysis"] == {} + assert result["overall_completeness"] == 0 + + def test_pattern_matching_picks_up_columns(self): + df = pd.DataFrame({ + "GTIN": ["614141000012"], + "Product Name": ["Marinara"], + "Brand Name": ["Acme"], + "Net Weight": ["12oz"], + }) + result = check_data_completeness(df) + assert "product_name" in result["field_analysis"] + assert "brand" in result["field_analysis"] + assert "weight" in result["field_analysis"] + + def test_completeness_pct_calculated(self): + df = pd.DataFrame({ + "Product Name": ["A", "B", "", None], + }) + result = check_data_completeness(df) + # 2 of 4 rows populated for product_name => 50% + assert result["field_analysis"]["product_name"]["completeness_pct"] == 50.0 + + def test_retailer_gap_analysis_flags_missing(self): + df = pd.DataFrame({"GTIN": ["614141000012"]}) + result = check_data_completeness(df) + # No retailer can be ready with only a GTIN column + for retailer, gaps in result["retailer_data_gaps"].items(): + assert gaps["ready"] is False + assert gaps["missing_fields"] + + +# ========================================================================= +# Report generators (smoke tests) +# ========================================================================= + +class TestCSVReport: + def test_csv_has_header_and_rows(self): + data = validate_batch(["614141000012", "61414100010"]) + csv_text = generate_csv_report(data) + lines = csv_text.strip().splitlines() + assert lines[0].startswith("Row,GTIN (Original),GTIN (Cleaned),") + assert len(lines) == 3 # header + 2 rows + + def test_csv_escapes_formula_injection(self): + # Simulate a malicious paste — leading '=' must be neutralized + data = validate_batch(["=cmd|'/c calc'!A1"]) + csv_text = generate_csv_report(data) + # The cell containing the malicious payload must be prefixed with ' + # so that Excel/Sheets treats it as literal text, not a formula + assert "\"'=cmd" in csv_text or "'=cmd" in csv_text + # And the bare formula prefix should never appear as a value + assert ",=cmd" not in csv_text + + def test_csv_handles_clean_input(self): + data = validate_batch(["614141000012"]) + csv_text = generate_csv_report(data) + # Row's status column should reflect a clean-ish result + assert "614141000012" in csv_text + + +class TestPDFReport: + def test_pdf_generates_for_typical_batch(self): + data = validate_batch([ + "614141000012", + "61414100010", + "000000000000", + "10614141000019", + ]) + buf = generate_pdf_report(data, company_name="Test Co.") + content = buf.getvalue() + assert content.startswith(b"%PDF-") + assert len(content) > 1000 # non-trivial document + + def test_pdf_handles_no_company_name(self): + data = validate_batch(["614141000012"]) + buf = generate_pdf_report(data, company_name="") + assert buf.getvalue().startswith(b"%PDF-") + + def test_pdf_escapes_markup_in_company_name(self): + # ReportLab parses inline markup; an unescaped '<' would corrupt + # or fail to render. The escape helper should prevent that. + data = validate_batch(["614141000012"]) + buf = generate_pdf_report(data, company_name="Bob & ") + assert buf.getvalue().startswith(b"%PDF-") + + +# ========================================================================= +# Sample data regression guard +# ========================================================================= + +class TestSampleData: + def test_sample_data_parses_and_validates(self): + from io import StringIO + from sample_data import SAMPLE_DATA + df = pd.read_csv(StringIO(SAMPLE_DATA.strip()), dtype=str) + gtins = df["GTIN"].dropna().tolist() + data = validate_batch(gtins) + assert data["summary"]["total_gtins"] == len(gtins) + # The sample is intentionally messy — there should be some critical + # issues and at least one duplicate + assert data["summary"]["critical_issues"] > 0 + assert data["summary"]["duplicate_groups"] >= 1 diff --git a/ui/__init__.py b/ui/__init__.py new file mode 100644 index 0000000..32f24d4 --- /dev/null +++ b/ui/__init__.py @@ -0,0 +1,6 @@ +"""UI helpers for the GTIN Validator Streamlit app. + +This package decomposes what used to be the monolithic app.py into +cohesive modules. The top-level app.py imports from here and stays +focused on page wiring. +""" diff --git a/ui/deep_analysis.py b/ui/deep_analysis.py new file mode 100644 index 0000000..d3833d8 --- /dev/null +++ b/ui/deep_analysis.py @@ -0,0 +1,345 @@ +"""Deep-analysis tabs: exec summary, fix plan, retailer readiness, +cost of inaction, GTIN-14 generator, data completeness.""" + +from __future__ import annotations + +from typing import Optional + +import pandas as pd +import streamlit as st + +from gtin_core import ( + check_data_completeness, + generate_executive_summary, + generate_fix_roadmap, + generate_gtin14_suggestions, +) + + +def render_deep_analysis( + validation_data: dict, + selected_retailer: str, + uploaded_df: Optional[pd.DataFrame], +) -> None: + st.markdown("---") + st.markdown("## 🔬 Deep Analysis") + st.markdown( + "Go beyond basic validation — understand what to fix first, " + "check retailer readiness, estimate costs, and track your progress." + ) + + tabs = st.tabs([ + "📝 Executive Summary", + "🗺️ Prioritized Fix Plan", + "🏪 Retailer Readiness", + "💰 Cost of Inaction", + "🔧 Case GTIN Generator", + "📊 Product Data Completeness", + ]) + tab_summary, tab_roadmap, tab_retailer, tab_cost, tab_gtin14, tab_completeness = tabs + + with tab_summary: + _render_executive_summary(validation_data) + with tab_roadmap: + _render_fix_roadmap(validation_data) + with tab_retailer: + _render_retailer_readiness(validation_data, selected_retailer) + with tab_cost: + _render_cost_of_inaction(validation_data) + with tab_gtin14: + _render_gtin14_generator(validation_data) + with tab_completeness: + _render_data_completeness(uploaded_df) + + +def _render_executive_summary(validation_data: dict) -> None: + st.markdown("### Executive Summary") + st.markdown( + "Copy this summary and send it to your team. " + "It's written in plain language — no jargon." + ) + exec_summary = generate_executive_summary(validation_data) + st.markdown( + f'
' + f'{exec_summary.replace(chr(10) + chr(10), "

")}' + f'
', + unsafe_allow_html=True, + ) + st.download_button( + label="📋 Copy Summary as Text", + data=exec_summary, + file_name="gtin_executive_summary.txt", + mime="text/plain", + use_container_width=True, + ) + + +def _render_fix_roadmap(validation_data: dict) -> None: + st.markdown("### Prioritized Fix Plan") + st.markdown( + "Issues ranked by **impact × effort**. Start at the top — " + "these are your fastest wins with the biggest payoff." + ) + roadmap = generate_fix_roadmap( + validation_data["results"], validation_data["hierarchy"], + ) + if not roadmap: + st.success("No issues to fix — your data is clean!") + return + + effort_colors = {"Low": "#28a745", "Medium": "#ffc107", "High": "#dc3545"} + impact_colors = {"High": "#dc3545", "Medium": "#ffc107", "Low": "#28a745"} + + for idx, item in enumerate(roadmap, 1): + effort_color = effort_colors.get(item["effort"], "#6c757d") + impact_color = impact_colors.get(item["impact"], "#6c757d") + + action_preview = item["action"][:80] + if len(item["action"]) > 80: + action_preview += "..." + plural = "s" if item["count"] != 1 else "" + + with st.expander( + f"Priority {idx}: {action_preview} ({item['count']} item{plural})" + ): + col_e, col_i, col_t = st.columns(3) + with col_e: + st.markdown( + f"**Effort:** {item['effort']}", + unsafe_allow_html=True, + ) + st.caption(item["effort_detail"]) + with col_i: + st.markdown( + f"**Impact:** {item['impact']}", + unsafe_allow_html=True, + ) + st.caption(item["impact_detail"]) + with col_t: + st.markdown("**Time estimate:**") + st.caption(item["time_estimate"]) + + st.markdown(f"**Full recommendation:** {item['action']}") + + +def _render_retailer_readiness( + validation_data: dict, selected_retailer: str, +) -> None: + st.markdown("### Retailer Submission Readiness") + st.markdown( + "Each retailer has specific GTIN requirements. " + "Here's how your data stacks up." + ) + checklists = validation_data["retailer_checklists"] + retailers_to_show = ( + {selected_retailer: checklists[selected_retailer]} + if selected_retailer != "All Retailers" + else checklists + ) + + for retailer_name, checklist in retailers_to_show.items(): + ready_class = "retailer-ready" if checklist["ready"] else "retailer-not-ready" + status_text = ( + "✅ READY" if checklist["ready"] + else f"❌ {checklist['passed']}/{checklist['total']} checks passed" + ) + st.markdown( + f""" +
+ {retailer_name} — {status_text}
+ {checklist['profile']['description']} +
+ """, + unsafe_allow_html=True, + ) + for check in checklist["checks"]: + icon = "✅" if check["passed"] else "❌" + st.markdown( + f"    {icon} {check['check']} — *{check['detail']}*" + ) + if checklist["profile"].get("notes"): + st.caption(checklist["profile"]["notes"]) + st.markdown("") + + +def _render_cost_of_inaction(validation_data: dict) -> None: + st.markdown("### Estimated Cost of Inaction") + st.markdown( + "These estimates are based on industry averages for specialty food brands " + "at similar scale. They're directional — meant to quantify the risk, not " + "predict exact costs." + ) + cost = validation_data["cost_estimate"] + if not cost: + st.info("No cost estimates available — no issues detected.") + return + + col_a, col_b = st.columns(2) + with col_a: + st.markdown( + f""" +
+
+ ${cost['annual_estimate_low']:,} – ${cost['annual_estimate_high']:,} +
+
Estimated annual cost of unresolved GTIN issues
+
+ """, + unsafe_allow_html=True, + ) + with col_b: + st.markdown( + f""" +
+
{cost['rework_hours']} hours/year
+
Manual rework from GTIN problems
+
+ """, + unsafe_allow_html=True, + ) + + st.markdown("#### Breakdown") + cost_df = pd.DataFrame([ + { + "Category": "Chargebacks from invalid GTINs", + "Low Estimate": f"${cost['chargeback_range'][0]:,}", + "High Estimate": f"${cost['chargeback_range'][1]:,}", + }, + { + "Category": f"Delayed launches ({cost['delayed_skus']} SKUs)", + "Low Estimate": f"${cost['delayed_launch_range'][0]:,}", + "High Estimate": f"${cost['delayed_launch_range'][1]:,}", + }, + { + "Category": f"Manual rework ({cost['rework_hours']} hrs)", + "Low Estimate": f"${cost['rework_cost']:,}", + "High Estimate": f"${cost['rework_cost']:,}", + }, + ]) + st.dataframe(cost_df, use_container_width=True, hide_index=True) + + if cost.get("growth_note"): + st.warning(f"📈 **Growth multiplier:** {cost['growth_note']}") + + +def _render_gtin14_generator(validation_data: dict) -> None: + st.markdown("### Case GTIN-14 Generator") + st.markdown( + "These are your unit-level GTINs that don't have a corresponding " + "case-level GTIN-14 in your file. Below are the GTIN-14s you'd need " + "to create for each packaging level." + ) + suggestions = generate_gtin14_suggestions( + validation_data["results"], validation_data["hierarchy"], + ) + if not suggestions: + st.success( + "All unit GTINs have matching case-level GTIN-14s, " + "or no valid unit GTINs were found to generate suggestions for." + ) + return + + st.markdown(f"**{len(suggestions)} unit GTIN(s)** need case-level GTIN-14s.") + for s in suggestions: + with st.expander(f"Row {s['row']}: {s['unit_gtin']} ({s['unit_type']})"): + rows = [ + { + "Indicator": str(ind), + "GTIN-14": info["gtin14"], + "Packaging Level": info["label"], + } + for ind, info in s["indicators"].items() + ] + st.dataframe(pd.DataFrame(rows), use_container_width=True, hide_index=True) + st.caption( + "Most commonly, indicator 1 = case. Copy the GTIN-14 you need " + "and add it to your product master." + ) + + +def _render_data_completeness(uploaded_df: Optional[pd.DataFrame]) -> None: + st.markdown("### Product Data Completeness") + if uploaded_df is None or len(uploaded_df.columns) <= 1: + st.info( + "Data completeness analysis is available when you upload a CSV file " + "with multiple columns (beyond just GTINs). Upload a product master " + "spreadsheet to see which fields are missing or incomplete." + ) + return + + st.markdown( + "Beyond GTINs, retailers require dozens of product attributes. " + "Here's what we found in your file." + ) + completeness = check_data_completeness(uploaded_df) + + if not completeness["field_analysis"]: + st.info( + "No standard product data fields detected beyond GTINs. " + "Upload a CSV with columns like Product Name, Brand, Weight, " + "Height, Width, Depth, etc. for a completeness analysis." + ) + return + + overall = completeness["overall_completeness"] + overall_color = ( + "#28a745" if overall >= 80 + else "#ffc107" if overall >= 50 + else "#dc3545" + ) + st.markdown( + f'
' + f'
' + f'{overall}%
' + f'
Overall Data Completeness
', + unsafe_allow_html=True, + ) + + st.markdown("#### Fields Found in Your File") + field_rows = [ + { + "Field": field_name.replace("_", " ").title(), + "Column": data["column_name"], + "Populated": f"{data['populated']}/{data['total_rows']}", + "% Rows Populated": f"{data['completeness_pct']}%", + } + for field_name, data in completeness["field_analysis"].items() + ] + st.dataframe(pd.DataFrame(field_rows), use_container_width=True, hide_index=True) + + if completeness["missing_important_fields"]: + st.markdown("#### Missing Important Fields") + st.warning( + "The following fields were not found in your file: **" + + ", ".join( + f.replace("_", " ").title() + for f in completeness["missing_important_fields"] + ) + + "**. Most retailers require these for item setup." + ) + + st.markdown("#### Retailer Data Readiness") + for retailer, gaps in completeness["retailer_data_gaps"].items(): + status = ( + "✅ READY" if gaps["ready"] + else f"❌ {gaps['present']}/{gaps['required']} fields present" + ) + with st.expander(f"{retailer} — {status}"): + if gaps["missing_fields"]: + st.markdown( + "**Missing:** " + + ", ".join( + f.replace("_", " ").title() + for f in gaps["missing_fields"] + ) + ) + if gaps["incomplete_fields"]: + st.markdown( + "**Incomplete (not all rows filled):** " + + ", ".join( + f.replace("_", " ").title() + for f in gaps["incomplete_fields"] + ) + ) + if gaps["ready"]: + st.success("All required fields present and complete.") diff --git a/ui/input_section.py b/ui/input_section.py new file mode 100644 index 0000000..8d67f38 --- /dev/null +++ b/ui/input_section.py @@ -0,0 +1,119 @@ +"""Input collection: paste, CSV upload, or sample data.""" + +from __future__ import annotations + +from io import StringIO +from typing import Optional + +import pandas as pd +import streamlit as st + +from sample_data import SAMPLE_DATA, SAMPLE_DESCRIPTION +from ui.state import MAX_GTINS_PER_BATCH + + +def render_input_section() -> tuple[list[str], Optional[pd.DataFrame]]: + """Render the input controls and return the parsed (gtins, dataframe). + + The dataframe is None unless the user uploaded a CSV or chose the + sample data path — it's used by the data-completeness section. + """ + input_method = st.radio( + "Choose input method:", + ["Paste GTINs", "Upload CSV", "Try sample data"], + horizontal=True, + ) + + if input_method == "Paste GTINs": + return _paste_input(), None + if input_method == "Upload CSV": + return _csv_upload_input() + return _sample_data_input() + + +# -- paste ------------------------------------------------------------------- + +def _paste_input() -> list[str]: + gtin_input = st.text_area( + "Paste your GTINs (one per line):", + height=200, + placeholder="614141000012\n614141000029\n614141000036\n...", + ) + if not gtin_input.strip(): + return [] + + parsed_lines = [ + line.strip() + for line in gtin_input.strip().split("\n") + if line.strip() + ] + if len(parsed_lines) > MAX_GTINS_PER_BATCH: + st.error( + f"Too many GTINs ({len(parsed_lines):,}). The current limit " + f"is {MAX_GTINS_PER_BATCH:,} per batch — please split your " + "list and validate it in chunks." + ) + return [] + return parsed_lines + + +# -- CSV upload -------------------------------------------------------------- + +def _csv_upload_input() -> tuple[list[str], Optional[pd.DataFrame]]: + uploaded_file = st.file_uploader( + "Upload a CSV file with a GTIN column:", + type=["csv"], + help="Your CSV should have a column containing GTINs. We'll auto-detect it.", + ) + if not uploaded_file: + return [], None + + try: + df = pd.read_csv(uploaded_file, dtype=str) + except (pd.errors.ParserError, UnicodeDecodeError, ValueError) as e: + st.error(f"Error reading CSV: {e}") + return [], None + except Exception as e: # noqa: BLE001 — surface to UI without crashing + st.error(f"Unexpected error reading CSV: {e}") + return [], None + + gtin_col = _detect_gtin_column(df) + if gtin_col is None: + gtin_col = st.selectbox( + "Which column contains GTINs?", df.columns.tolist(), + ) + else: + st.info(f"Auto-detected GTIN column: **{gtin_col}**") + + parsed_lines = df[gtin_col].dropna().tolist() + if len(parsed_lines) > MAX_GTINS_PER_BATCH: + st.error( + f"Too many GTINs ({len(parsed_lines):,}). The current " + f"limit is {MAX_GTINS_PER_BATCH:,} per batch — please " + "split your file and validate it in chunks." + ) + return [], df + + st.success(f"Loaded {len(parsed_lines)} GTINs from '{gtin_col}'") + return parsed_lines, df + + +_GTIN_COL_HINTS = ("gtin", "upc", "ean", "barcode", "code") + + +def _detect_gtin_column(df: pd.DataFrame) -> Optional[str]: + for col in df.columns: + if any(hint in col.lower() for hint in _GTIN_COL_HINTS): + return col + return None + + +# -- sample data ------------------------------------------------------------- + +def _sample_data_input() -> tuple[list[str], pd.DataFrame]: + st.markdown(SAMPLE_DESCRIPTION) + sample_df = pd.read_csv(StringIO(SAMPLE_DATA.strip()), dtype=str) + st.dataframe(sample_df, use_container_width=True, height=300) + gtins = sample_df["GTIN"].dropna().tolist() + st.info(f"Loaded {len(gtins)} sample GTINs") + return gtins, sample_df diff --git a/ui/results.py b/ui/results.py new file mode 100644 index 0000000..aa72261 --- /dev/null +++ b/ui/results.py @@ -0,0 +1,277 @@ +"""Validation results: score card, summary stats, downloads, results tabs.""" + +from __future__ import annotations + +import pandas as pd +import streamlit as st + +from csv_report import generate_csv_report +from gtin_core import Severity, generate_before_after +from pdf_report import generate_pdf_report +from ui.state import ( + KEY_CSV_CACHE, KEY_PDF_CACHE, KEY_PDF_COMPANY, KEY_PDF_ERROR, +) + + +# -- score + summary --------------------------------------------------------- + +def render_score_card(score: dict) -> None: + score_color = ( + "#28a745" if score["score"] >= 75 + else "#ffc107" if score["score"] >= 50 + else "#dc3545" + ) + st.markdown( + f""" +
+
{score["score"]}
+
Grade: {score["grade"]}
+
{score["interpretation"]}
+
+ """, + unsafe_allow_html=True, + ) + + +def render_summary_stats(summary: dict) -> None: + cols = st.columns(4) + cards = [ + ("stat-card", summary["total_gtins"], "Total GTINs"), + ("stat-card stat-critical", summary["critical_issues"], "Critical Issues"), + ("stat-card stat-warning", summary["warnings"], "Warnings"), + ("stat-card stat-clean", summary["clean"], "Clean"), + ] + for col, (cls, value, label) in zip(cols, cards): + with col: + st.markdown( + f""" +
+
{value}
+
{label}
+
+ """, + unsafe_allow_html=True, + ) + + +# -- downloads --------------------------------------------------------------- + +def render_download_buttons(validation_data: dict, company_name: str) -> None: + st.markdown("### 📥 Download Validation Reports") + col_csv, col_pdf = st.columns(2) + filename_base = company_name.replace(" ", "_") if company_name else "gtin_validation" + + with col_csv: + _render_csv_download(validation_data, filename_base) + with col_pdf: + _render_pdf_download(validation_data, company_name, filename_base) + + +def _render_csv_download(validation_data: dict, filename_base: str) -> None: + st.markdown("**📄 CSV Report — Raw Data**") + st.markdown( + '

' + 'Row-by-row validation results in spreadsheet format. ' + 'Includes each GTIN, its status, issue codes, and corrected values. ' + 'Best for importing into Excel or your own systems for further analysis.' + '

', + unsafe_allow_html=True, + ) + if KEY_CSV_CACHE not in st.session_state: + st.session_state[KEY_CSV_CACHE] = generate_csv_report(validation_data) + st.download_button( + label="📄 Download CSV Report", + data=st.session_state[KEY_CSV_CACHE], + file_name=f"{filename_base}_report.csv", + mime="text/csv", + use_container_width=True, + ) + + +def _render_pdf_download( + validation_data: dict, company_name: str, filename_base: str, +) -> None: + st.markdown("**📑 PDF Report — Full Diagnostic**") + st.markdown( + '

' + 'Branded, professional report with readiness score, retailer-specific ' + 'checklists, cost-of-inaction estimates, and prioritized issue detail. ' + 'Designed to hand directly to your operations team, broker, or trading partner coordinator.' + '

', + unsafe_allow_html=True, + ) + + pdf_cache_stale = ( + KEY_PDF_CACHE not in st.session_state + or st.session_state.get(KEY_PDF_COMPANY) != company_name + ) + if pdf_cache_stale: + try: + st.session_state[KEY_PDF_CACHE] = generate_pdf_report( + validation_data, company_name, + ) + st.session_state[KEY_PDF_COMPANY] = company_name + st.session_state.pop(KEY_PDF_ERROR, None) + except Exception as e: # noqa: BLE001 + st.session_state[KEY_PDF_CACHE] = None + st.session_state[KEY_PDF_ERROR] = str(e) + + if st.session_state.get(KEY_PDF_ERROR): + st.error(f"PDF generation error: {st.session_state[KEY_PDF_ERROR]}") + return + + st.download_button( + label="📑 Download PDF Report", + data=st.session_state[KEY_PDF_CACHE], + file_name=f"{filename_base}_report.pdf", + mime="application/pdf", + use_container_width=True, + ) + + +# -- results tabs ------------------------------------------------------------ + +def render_results_tabs(validation_data: dict) -> None: + """Render the four validation-result tabs.""" + results = validation_data["results"] + hierarchy = validation_data["hierarchy"] + + tab_issues, tab_detail, tab_check_digit_fixes, tab_item_detail = st.tabs([ + "📋 Issues by Severity", + "🔍 Full Item Detail", + "✏️ Check Digit Corrections", + "📦 Packaging Hierarchy", + ]) + with tab_issues: + _render_issues_by_severity(results) + with tab_detail: + _render_full_item_detail(results) + with tab_check_digit_fixes: + _render_check_digit_fixes(results) + with tab_item_detail: + _render_packaging_hierarchy(hierarchy) + + +def _render_issues_by_severity(results) -> None: + st.markdown("### Issues by Severity") + critical_items = [r for r in results if r.has_critical] + warning_items = [ + r for r in results if r.has_warning and not r.has_critical + ] + info_items = [ + r for r in results + if r.issues and not r.has_critical and not r.has_warning + ] + + if critical_items: + st.markdown( + 'CRITICAL — ' + 'These GTINs will be **rejected** by retailers.', + unsafe_allow_html=True, + ) + for r in critical_items: + with st.expander(f"Row {r.row_number}: {r.raw_input}"): + for issue in r.issues: + if issue.severity == Severity.CRITICAL: + st.error(f"**{issue.message}**") + st.markdown(f"**Fix:** {issue.recommendation}") + st.markdown(f"**Retailer impact:** {issue.retailer_impact}") + st.markdown("---") + + if warning_items: + st.markdown( + 'WARNING — ' + 'These GTINs may cause problems.', + unsafe_allow_html=True, + ) + for r in warning_items: + with st.expander(f"Row {r.row_number}: {r.raw_input}"): + for issue in r.issues: + st.warning(f"**{issue.message}**") + st.markdown(f"**Fix:** {issue.recommendation}") + st.markdown(f"**Retailer impact:** {issue.retailer_impact}") + + if info_items: + st.markdown( + 'INFO — Best practice notes.', + unsafe_allow_html=True, + ) + for r in info_items: + with st.expander(f"Row {r.row_number}: {r.raw_input}"): + for issue in r.issues: + st.info(f"{issue.message}") + + if not critical_items and not warning_items and not info_items: + st.success("🎉 All GTINs passed validation with no issues!") + + +def _render_full_item_detail(results) -> None: + st.markdown("### Full Item Detail") + detail_rows = [] + for r in results: + status = ( + "✅ Clean" if not r.issues + else "❌ Critical" if r.has_critical + else "⚠️ Warning" if r.has_warning + else "ℹ️ Info" + ) + detail_rows.append({ + "Row": r.row_number, + "GTIN": r.raw_input, + "Type": r.gtin_type.value, + "Status": status, + "Issues": len(r.issues), + "Corrected": r.corrected_value or "", + }) + st.dataframe(pd.DataFrame(detail_rows), use_container_width=True, hide_index=True) + + +def _render_check_digit_fixes(results) -> None: + st.markdown("### Check Digit Corrections") + st.markdown( + "These GTINs have incorrect check digits. The corrected values are shown below. " + "**Important:** always verify corrections against your original barcode or GS1 " + "registration before updating your product master." + ) + before_after = generate_before_after(results) + if before_after: + ba_df = pd.DataFrame(before_after) + ba_df.columns = ["Row", "Current (Before)", "Corrected (After)", "Issue"] + st.dataframe(ba_df, use_container_width=True, hide_index=True) + else: + st.success("No check digit corrections needed — all check digits are valid.") + + +def _render_packaging_hierarchy(hierarchy: dict) -> None: + st.markdown("### Packaging Hierarchy Analysis") + st.markdown( + "Retailers like Walmart require GTINs at every packaging level — " + "each, inner pack, case, and pallet. This analysis checks whether " + "your case-level GTIN-14s match up with unit-level GTINs." + ) + + if hierarchy["matched_pairs"]: + st.markdown("#### ✅ Matched unit → case pairs") + pairs_df = pd.DataFrame(hierarchy["matched_pairs"]) + pairs_df.columns = ["Case GTIN", "Case Row", "Unit GTIN", "Unit Row", "Indicator"] + st.dataframe(pairs_df, use_container_width=True, hide_index=True) + + if hierarchy["orphan_cases"]: + st.markdown("#### ⚠️ Case GTINs without matching unit GTINs") + for r in hierarchy["orphan_cases"]: + st.warning(f"Row {r.row_number}: **{r.cleaned}** — no matching unit GTIN found") + + if hierarchy["units_without_cases"]: + st.markdown("#### 📦 Unit GTINs without case-level GTINs") + st.caption( + "These items don't have a corresponding GTIN-14 for case/shipping identification. " + "If you ship these to retailers in cases, you'll need case GTINs." + ) + for r in hierarchy["units_without_cases"]: + st.info(f"Row {r.row_number}: **{r.cleaned}** ({r.gtin_type.value})") + + if not hierarchy["matched_pairs"] and not hierarchy["orphan_cases"]: + st.info( + "No GTIN-14 case-level codes found in your data. " + "If you ship to Walmart or Costco, you'll likely need case GTINs (GTIN-14 with indicator digits 1-8)." + ) diff --git a/ui/state.py b/ui/state.py new file mode 100644 index 0000000..d9d8fba --- /dev/null +++ b/ui/state.py @@ -0,0 +1,59 @@ +"""Session-state key constants and helpers. + +Centralizing these here means the rest of the UI never deals with raw +string keys, and reset_session() can no longer accidentally wipe +unrelated keys. +""" + +from __future__ import annotations + +import streamlit as st + + +# Hard cap on rows we will validate from any input source. Keeps Streamlit +# responsive when someone pastes (or uploads) a huge list by accident. +MAX_GTINS_PER_BATCH = 50_000 + + +# -- session_state keys ------------------------------------------------------ + +KEY_GTINS = "gtins_to_validate" +KEY_DF = "uploaded_df" +KEY_VALIDATED = "validated" +KEY_VALIDATION_CACHE = "validation_data_cache" +KEY_CSV_CACHE = "csv_report_cache" +KEY_PDF_CACHE = "pdf_report_cache" +KEY_PDF_COMPANY = "pdf_report_company_name" +KEY_PDF_ERROR = "pdf_report_error" + +# Keys this UI owns. reset_session() only clears these — anything else +# (e.g. Streamlit-internal widget state) is left alone. +_OWNED_KEYS = ( + KEY_GTINS, + KEY_DF, + KEY_VALIDATED, + KEY_VALIDATION_CACHE, + KEY_CSV_CACHE, + KEY_PDF_CACHE, + KEY_PDF_COMPANY, + KEY_PDF_ERROR, +) + + +# -- helpers ----------------------------------------------------------------- + +def reset_session() -> None: + """Clear every key this UI owns. Used by the Reset button.""" + for key in _OWNED_KEYS: + st.session_state.pop(key, None) + + +def invalidate_report_caches() -> None: + """Drop the derived CSV / PDF caches so they regenerate on next use. + + Called whenever validate_batch is re-run with fresh input. + """ + st.session_state.pop(KEY_CSV_CACHE, None) + st.session_state.pop(KEY_PDF_CACHE, None) + st.session_state.pop(KEY_PDF_COMPANY, None) + st.session_state.pop(KEY_PDF_ERROR, None) diff --git a/ui/styles.css b/ui/styles.css new file mode 100644 index 0000000..59fd2bf --- /dev/null +++ b/ui/styles.css @@ -0,0 +1,181 @@ +/* + * GTIN Validator — application styles. + * + * Loaded once at startup by ui.styles.inject_css(). Editing this file + * does NOT require touching any Python. + */ + +:root { + --bg-primary: #eaecee; + --bg-secondary: #e0e2e5; + --bg-card: #f5f5f5; + --bg-input: #ffffff; + --text-primary: #1a1a2e; + --text-secondary: #4a4a5a; + --text-muted: #6c757d; + --border-color: #d0d3d8; + --stat-card-bg: #f0f1f3; + --stat-card-border: #d0d3d8; + --retailer-card-bg: #f5f5f5; + --cost-card-bg: linear-gradient(135deg, #fff3cd 0%, #ffeeba 100%); + --cost-card-border: #ffc107; + --cost-number-color: #856404; + --security-bg: #e8f5e9; + --security-border: #c3e6cb; +} + +@import url('https://fonts.googleapis.com/css2?family=DM+Sans:wght@400;500;700&display=swap'); + +.stApp { + background-color: #eaecee !important; + font-family: 'DM Sans', sans-serif; +} +[data-testid="stSidebar"] { background-color: #e0e2e5 !important; } +.stTabs [data-baseweb="tab"] { color: #1a1a2e !important; } +.stTabs [data-baseweb="tab"][aria-selected="true"] { + color: #1a1a2e !important; + font-weight: 600; +} +.stTextInput input, .stTextArea textarea { + background-color: #ffffff !important; + color: #1a1a2e !important; + border-color: #d0d3d8 !important; +} +[data-baseweb="select"], +[data-baseweb="select"] div, +[data-baseweb="select"] span { + color: #1a1a2e !important; +} + +/* Score card */ +.score-card { + background: linear-gradient(135deg, #1a1a2e 0%, #16213e 100%); + border-radius: 16px; + padding: 2rem; + text-align: center; + color: white; + margin-bottom: 1rem; +} +.score-number { + font-size: 4rem; + font-weight: 700; + line-height: 1; + margin-bottom: 0.25rem; + color: white !important; +} +.score-grade { + font-size: 1.2rem; + opacity: 0.8; + margin-bottom: 0.5rem; + color: white !important; +} +.score-interp { + font-size: 0.95rem; + opacity: 0.7; + color: white !important; +} + +/* Stat cards */ +.stat-row { + display: flex; + gap: 1rem; + margin-bottom: 1rem; +} +.stat-card { + background: var(--stat-card-bg); + border-radius: 12px; + padding: 1.25rem; + flex: 1; + text-align: center; + border: 1px solid var(--stat-card-border); +} +.stat-number { + font-size: 2rem; + font-weight: 700; + color: var(--text-primary) !important; +} +.stat-label { + font-size: 0.85rem; + color: var(--text-muted) !important; + margin-top: 0.25rem; +} +.stat-critical .stat-number { color: #dc3545 !important; } +.stat-warning .stat-number { color: #ffc107 !important; } +.stat-clean .stat-number { color: #28a745 !important; } + +/* Retailer checklist */ +.retailer-card { + background: var(--retailer-card-bg); + border: 1px solid var(--border-color); + border-radius: 12px; + padding: 1.25rem; + margin-bottom: 0.75rem; +} +.retailer-ready { border-left: 4px solid #28a745; } +.retailer-not-ready { border-left: 4px solid #dc3545; } +.check-pass { color: #28a745 !important; } +.check-fail { color: #dc3545 !important; } + +/* Cost card */ +.cost-card { + background: var(--cost-card-bg); + border: 1px solid var(--cost-card-border); + border-radius: 12px; + padding: 1.5rem; + margin-bottom: 1rem; +} +.cost-number { + font-size: 1.5rem; + font-weight: 700; + color: var(--cost-number-color) !important; +} + +/* Issue badges */ +.badge-critical, .badge-critical * { + background: #dc3545; + color: #ffffff !important; + padding: 2px 8px; + border-radius: 4px; + font-size: 0.75rem; + font-weight: 600; +} +.badge-warning, .badge-warning * { + background: #ffc107; + color: #1a1a2e !important; + padding: 2px 8px; + border-radius: 4px; + font-size: 0.75rem; + font-weight: 600; +} +.badge-info, .badge-info * { + background: #17a2b8; + color: #ffffff !important; + padding: 2px 8px; + border-radius: 4px; + font-size: 0.75rem; + font-weight: 600; +} + +/* Security disclaimer */ +.security-box { + background: var(--security-bg); + border: 1px solid var(--security-border); + border-radius: 8px; + padding: 1.25rem; + margin-top: 1rem; +} +.security-box-compact { + background: var(--security-bg); + border: 1px solid var(--security-border); + border-radius: 8px; + padding: 1rem; + margin-top: 1rem; +} + +/* Hide streamlit branding */ +#MainMenu { visibility: hidden; } +footer { visibility: hidden; } + +/* Cleaner tabs */ +.stTabs [data-baseweb="tab-list"] { gap: 2px; } +.stTabs [data-baseweb="tab"] { padding: 10px 20px; } diff --git a/ui/styles.py b/ui/styles.py new file mode 100644 index 0000000..ff1ef76 --- /dev/null +++ b/ui/styles.py @@ -0,0 +1,18 @@ +"""CSS injection for the Streamlit app.""" + +from pathlib import Path + +import streamlit as st + +_STYLES_PATH = Path(__file__).with_name("styles.css") + + +def inject_css() -> None: + """Inject the application stylesheet exactly once per session. + + Streamlit re-runs the entire script on every interaction, so we + guard with a session-state flag to avoid emitting the ", unsafe_allow_html=True)