diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..89af175 --- /dev/null +++ b/.gitignore @@ -0,0 +1,17 @@ +graphrag-ollama-config/__pycache__/* +graphrag-ollama-config/cache/* +graphrag-ollama-config/graphrag-ollama/* +graphrag-ollama-config/output/* +mcp/client/node_modules/* +veritasgraph/__pycache__/* +mcp/veritas-mcp-server/__pycache__/* +mcp/client/.next/* +tests/__pycache__/* +graphrag-ollama-config/input/youtube_EY_CAP_Discover_How_to_Analyze_your_Climate_Risks__a5f1e91a.txt +# scripts/*.sh + +# Environment files with secrets +.env +*.env +!*.env.example +!*.env.template diff --git a/ICASF 2025 - Appreciation Certificate.pdf b/ICASF 2025 - Appreciation Certificate.pdf deleted file mode 100644 index 2ac0cca..0000000 Binary files a/ICASF 2025 - Appreciation Certificate.pdf and /dev/null differ diff --git a/LICENSE b/LICENSE deleted file mode 100644 index 1718e73..0000000 --- a/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2025 BIBIN PRATHAP - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/MANIFEST.in b/MANIFEST.in deleted file mode 100644 index 915ce5e..0000000 --- a/MANIFEST.in +++ /dev/null @@ -1,37 +0,0 @@ -# Include essential files -include LICENSE -include README.md -include pyproject.toml - -# Include the package -recursive-include veritasgraph *.py -recursive-include veritasgraph *.pyi -include veritasgraph/py.typed - -# Exclude development and deployment files -exclude .gitignore -exclude .pre-commit-config.yaml -exclude Makefile - -# Exclude directories not needed in distribution -prune .git -prune .github -prune .venv -prune .vscode -prune docker -prune docs -prune finetune -prune graphrag-ollama-config -prune scripts -prune output -prune tests -prune assets - - -# Exclude common temp files -global-exclude *.py[cod] -global-exclude __pycache__ -global-exclude *.so -global-exclude .DS_Store -global-exclude *.egg-info - diff --git a/README_files/colorschememapping.xml b/README_files/colorschememapping.xml deleted file mode 100644 index 6a0069c..0000000 --- a/README_files/colorschememapping.xml +++ /dev/null @@ -1,2 +0,0 @@ - - \ No newline at end of file diff --git a/README_files/filelist.xml b/README_files/filelist.xml deleted file mode 100644 index 070fdda..0000000 --- a/README_files/filelist.xml +++ /dev/null @@ -1,24 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - \ No newline at end of file diff --git a/README_files/image001.png b/README_files/image001.png deleted file mode 100644 index 0a40ce3..0000000 Binary files a/README_files/image001.png and /dev/null differ diff --git a/README_files/image002.png b/README_files/image002.png deleted file mode 100644 index bccc52a..0000000 Binary files a/README_files/image002.png and /dev/null differ diff --git a/README_files/image003.png b/README_files/image003.png deleted file mode 100644 index d062629..0000000 Binary files a/README_files/image003.png and /dev/null differ diff --git a/README_files/image004.png b/README_files/image004.png deleted file mode 100644 index ba078ea..0000000 Binary files a/README_files/image004.png and /dev/null differ diff --git a/README_files/image005.png b/README_files/image005.png deleted file mode 100644 index 0071d5c..0000000 Binary files a/README_files/image005.png and /dev/null differ diff --git a/README_files/image006.png b/README_files/image006.png deleted file mode 100644 index bda4b91..0000000 Binary files a/README_files/image006.png and /dev/null differ diff --git a/README_files/image007.png b/README_files/image007.png deleted file mode 100644 index f24278f..0000000 Binary files a/README_files/image007.png and /dev/null differ diff --git a/README_files/image008.png b/README_files/image008.png deleted file mode 100644 index 59265d8..0000000 Binary files a/README_files/image008.png and /dev/null differ diff --git a/README_files/image009.png b/README_files/image009.png deleted file mode 100644 index d202823..0000000 Binary files a/README_files/image009.png and /dev/null differ diff --git a/README_files/image010.png b/README_files/image010.png deleted file mode 100644 index b6a0ed2..0000000 Binary files a/README_files/image010.png and /dev/null differ diff --git a/README_files/image011.png b/README_files/image011.png deleted file mode 100644 index d45c6d3..0000000 Binary files a/README_files/image011.png and /dev/null differ diff --git a/README_files/image012.png b/README_files/image012.png deleted file mode 100644 index b5da921..0000000 Binary files a/README_files/image012.png and /dev/null differ diff --git a/README_files/image013.png b/README_files/image013.png deleted file mode 100644 index a537b20..0000000 Binary files a/README_files/image013.png and /dev/null differ diff --git a/README_files/image014.png b/README_files/image014.png deleted file mode 100644 index 880ad97..0000000 Binary files a/README_files/image014.png and /dev/null differ diff --git a/README_files/image015.png b/README_files/image015.png deleted file mode 100644 index cdf047d..0000000 Binary files a/README_files/image015.png and /dev/null differ diff --git a/README_files/image016.png b/README_files/image016.png deleted file mode 100644 index b707452..0000000 Binary files a/README_files/image016.png and /dev/null differ diff --git a/README_files/image017.png b/README_files/image017.png deleted file mode 100644 index 701a6bb..0000000 Binary files a/README_files/image017.png and /dev/null differ diff --git a/README_files/image018.png b/README_files/image018.png deleted file mode 100644 index e3984f0..0000000 Binary files a/README_files/image018.png and /dev/null differ diff --git a/README_files/image019.png b/README_files/image019.png deleted file mode 100644 index bccc52a..0000000 Binary files a/README_files/image019.png and /dev/null differ diff --git a/README_files/image020.png b/README_files/image020.png deleted file mode 100644 index 6f1f100..0000000 Binary files a/README_files/image020.png and /dev/null differ diff --git a/README_files/image021.png b/README_files/image021.png deleted file mode 100644 index 875f3e2..0000000 Binary files a/README_files/image021.png and /dev/null differ diff --git a/README_files/image022.png b/README_files/image022.png deleted file mode 100644 index 512fef6..0000000 Binary files a/README_files/image022.png and /dev/null differ diff --git a/README_files/image023.png b/README_files/image023.png deleted file mode 100644 index 9b6be6e..0000000 Binary files a/README_files/image023.png and /dev/null differ diff --git a/README_files/image024.png b/README_files/image024.png deleted file mode 100644 index 59265d8..0000000 Binary files a/README_files/image024.png and /dev/null differ diff --git a/README_files/image025.png b/README_files/image025.png deleted file mode 100644 index b6a0ed2..0000000 Binary files a/README_files/image025.png and /dev/null differ diff --git a/README_files/image026.png b/README_files/image026.png deleted file mode 100644 index b5da921..0000000 Binary files a/README_files/image026.png and /dev/null differ diff --git a/README_files/image027.png b/README_files/image027.png deleted file mode 100644 index 880ad97..0000000 Binary files a/README_files/image027.png and /dev/null differ diff --git a/README_files/image028.png b/README_files/image028.png deleted file mode 100644 index b707452..0000000 Binary files a/README_files/image028.png and /dev/null differ diff --git a/README_files/image029.png b/README_files/image029.png deleted file mode 100644 index fb684cf..0000000 Binary files a/README_files/image029.png and /dev/null differ diff --git a/README_files/image030.png b/README_files/image030.png deleted file mode 100644 index 18f039a..0000000 Binary files a/README_files/image030.png and /dev/null differ diff --git a/README_files/themedata.thmx b/README_files/themedata.thmx deleted file mode 100644 index d641fac..0000000 Binary files a/README_files/themedata.thmx and /dev/null differ diff --git a/VeritasGraph-Sovereign-GraphRAG-for-the-Enterprise.pptx b/VeritasGraph-Sovereign-GraphRAG-for-the-Enterprise.pptx deleted file mode 100644 index a23abce..0000000 Binary files a/VeritasGraph-Sovereign-GraphRAG-for-the-Enterprise.pptx and /dev/null differ diff --git a/dist/veritasgraph-0.3.0-py3-none-any.whl b/dist/veritasgraph-0.3.0-py3-none-any.whl new file mode 100644 index 0000000..8c83e49 Binary files /dev/null and b/dist/veritasgraph-0.3.0-py3-none-any.whl differ diff --git a/dist/veritasgraph-0.3.0.tar.gz b/dist/veritasgraph-0.3.0.tar.gz new file mode 100644 index 0000000..221996d Binary files /dev/null and b/dist/veritasgraph-0.3.0.tar.gz differ diff --git a/docker/five-minute-magic-onboarding/.env b/docker/five-minute-magic-onboarding/.env deleted file mode 100644 index fbbeab1..0000000 --- a/docker/five-minute-magic-onboarding/.env +++ /dev/null @@ -1,11 +0,0 @@ -DATABASE_URL=neo4j://neo4j:7687 -DATABASE_USER=neo4j -DATABASE_PASSWORD=your_password_here - -NEO4J_USER=${DATABASE_USER} -NEO4J_PASS=${DATABASE_PASSWORD} - -OLLAMA_MODEL=llama3.1 -OLLAMA_EMBED_MODEL=nomic-embed-text - -SHARED_NETWORK=magic_network \ No newline at end of file diff --git a/docker/five-minute-magic-onboarding/README.md b/docker/five-minute-magic-onboarding/README.md deleted file mode 100644 index acd4ded..0000000 --- a/docker/five-minute-magic-onboarding/README.md +++ /dev/null @@ -1,58 +0,0 @@ -## Five-Minute Magic Onboarding - -Zero-config Docker Compose stack that spins up the VeritasGraph app, Neo4j, and Ollama on a shared bridge network. - -### Prerequisites - -- Docker Engine 24+ -- Docker Compose plugin 2.20+ -- At least 16 GB RAM (Ollama + Neo4j + app) - -### Structure - -``` -docker/five-minute-magic-onboarding/ -├── .env # Credentials and shared config -├── docker-compose.yaml # Orchestrates ollama + neo4j + app -├── app/ # Gradio UI container build context -├── neo4j/conf/neo4j.conf # Optional overrides -└── ollama/Modelfile # Defines llama3.1-12k custom model -``` - -### Configure - -1. Copy `.env` and set credentials: - ```env - DATABASE_USER=neo4j - DATABASE_PASSWORD=your_password_here - NEO4J_USER=${DATABASE_USER} - NEO4J_PASS=${DATABASE_PASSWORD} - ``` -2. (Optional) adjust `ollama/Modelfile` for different LLMs. - -### Run - -```bash -cd docker/five-minute-magic-onboarding -docker compose up --build -``` - -What happens: -- **ollama** container loads `llama3.1-12k` if missing and exposes `11434`. -- **neo4j** container boots with auth from `.env` and exposes `7474/7687`. -- **app** container mounts `../graphrag-ollama-config`, injects env vars, and serves Gradio UI on `7860`. - -### Verify - -- Ollama tags: `curl http://localhost:11434/api/tags` -- Neo4j browser: http://localhost:7474/ -- Gradio UI: http://127.0.0.1:7860/ - -### Stop & Cleanup - -```bash -docker compose down # stop containers -docker compose down -v # stop and remove volumes (models/data) -``` - -This workflow delivers a reproducible, single-command demo for VeritasGraph onboarding. diff --git a/docker/five-minute-magic-onboarding/app/Dockerfile b/docker/five-minute-magic-onboarding/app/Dockerfile deleted file mode 100644 index 456c745..0000000 --- a/docker/five-minute-magic-onboarding/app/Dockerfile +++ /dev/null @@ -1,11 +0,0 @@ -FROM python:3.9-slim - -WORKDIR /app - -COPY requirements.txt . - -RUN pip install --no-cache-dir -r requirements.txt - -COPY src/ . - -CMD ["python", "main.py"] \ No newline at end of file diff --git a/docker/five-minute-magic-onboarding/app/requirements.txt b/docker/five-minute-magic-onboarding/app/requirements.txt deleted file mode 100644 index 5d5a893..0000000 --- a/docker/five-minute-magic-onboarding/app/requirements.txt +++ /dev/null @@ -1,5 +0,0 @@ -Flask==2.1.1 -neo4j==4.4.6 -requests==2.26.0 -ollama==0.1.0 -gunicorn==20.1.0 \ No newline at end of file diff --git a/docker/five-minute-magic-onboarding/app/src/main.py b/docker/five-minute-magic-onboarding/app/src/main.py deleted file mode 100644 index 764c2bc..0000000 --- a/docker/five-minute-magic-onboarding/app/src/main.py +++ /dev/null @@ -1,10 +0,0 @@ -from flask import Flask, jsonify - -app = Flask(__name__) - -@app.route('/') -def home(): - return jsonify(message="Welcome to the Five-Minute Magic Onboarding!") - -if __name__ == '__main__': - app.run(host='0.0.0.0', port=5000) \ No newline at end of file diff --git a/docker/five-minute-magic-onboarding/docker-compose.yaml b/docker/five-minute-magic-onboarding/docker-compose.yaml deleted file mode 100644 index 42911e5..0000000 --- a/docker/five-minute-magic-onboarding/docker-compose.yaml +++ /dev/null @@ -1,80 +0,0 @@ -version: "3.9" - -services: - ollama: - image: ollama/ollama:latest - container_name: vg_ollama - restart: unless-stopped - ports: - - "11434:11434" - volumes: - - ollama_models:/root/.ollama - - ./ollama/Modelfile:/Modelfile - command: > - /bin/sh -c " - ollama serve & - sleep 3 && - if ! ollama list | grep -q 'llama3.1-12k'; then - ollama create llama3.1-12k -f /Modelfile; - fi && - tail -f /dev/null - " - healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:11434/api/tags"] - interval: 30s - timeout: 5s - retries: 5 - networks: - - veritasgraph - - neo4j: - image: neo4j:5.23 - container_name: vg_neo4j - restart: unless-stopped - environment: - NEO4J_AUTH: ${NEO4J_USER}/${NEO4J_PASS} - NEO4J_dbms_security_auth__enabled: "true" - ports: - - "7474:7474" - - "7687:7687" - volumes: - - neo4j_data:/data - - ./neo4j/conf/neo4j.conf:/var/lib/neo4j/conf/neo4j.conf - networks: - - veritasgraph - depends_on: - - ollama - - app: - build: - context: ./app - dockerfile: Dockerfile - container_name: vg_app - restart: unless-stopped - env_file: - - .env - environment: - GRAPHRAG_LLM_API_BASE: http://ollama:11434 - GRAPHRAG_EMBEDDING_API_BASE: http://ollama:11434 - NEO4J_URI: bolt://neo4j:7687 - NEO4J_USER: ${NEO4J_USER} - NEO4J_PASS: ${NEO4J_PASS} - volumes: - - ../graphrag-ollama-config:/workspace - ports: - - "7860:7860" - depends_on: - ollama: - condition: service_healthy - neo4j: - condition: service_started - networks: - - veritasgraph - -volumes: - ollama_models: - neo4j_data: - -networks: - veritasgraph: - driver: bridge \ No newline at end of file diff --git a/docker/five-minute-magic-onboarding/ollama/Modelfile b/docker/five-minute-magic-onboarding/ollama/Modelfile deleted file mode 100644 index 63b31ea..0000000 --- a/docker/five-minute-magic-onboarding/ollama/Modelfile +++ /dev/null @@ -1,6 +0,0 @@ -model: "llama3.1" -context_length: 12288 -embedding_model: "nomic-embed-text" -output_dir: "./output" -cache_dir: "./cache" -log_level: "info" \ No newline at end of file diff --git a/docker/five-minute-magic-onboarding/neo4j/conf/neo4j.conf b/docs/.nojekyll similarity index 100% rename from docker/five-minute-magic-onboarding/neo4j/conf/neo4j.conf rename to docs/.nojekyll diff --git a/docs/demo/index.html b/docs/demo/index.html index 8df73d8..b5b3473 100644 --- a/docs/demo/index.html +++ b/docs/demo/index.html @@ -3,8 +3,8 @@ - - + + VeritasGraph Live Demo - Redirecting... @@ -62,16 +48,14 @@

🔍 VeritasGraph Demo

Redirecting to live demo...

- Click here if not redirected + Click here if not redirected
Status: Server is online
- Last updated: 2026-02-03 05:41 UTC + Last updated: 2026-03-22 07:23 UTC
diff --git a/finetune/data.jsonl b/finetune/data.jsonl deleted file mode 100644 index 37c19d2..0000000 --- a/finetune/data.jsonl +++ /dev/null @@ -1,10 +0,0 @@ -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "What are the main features of the Customer Analytics Platform?"}, {"from": "assistant", "value": "1. Real-time customer behavior tracking\n2. Demographic segmentation\n3. Purchase pattern analysis\n4. Predictive churn modeling\n5. Custom report generation\n6. Integration with CRM systems"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "How do I run the unit tests for the authentication module?"}, {"from": "assistant", "value": "To run the unit tests for the authentication module:\n\n1. Navigate to the project root directory\n2. Activate your virtual environment\n3. Run the command: `pytest tests/auth/`\n4. For coverage report add: `--cov=app/auth`"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "Explain the database schema for the inventory management system"}, {"from": "assistant", "value": "The inventory management system uses 5 main tables:\n\n- Products: id, name, description, SKU, price, created_at\n- Categories: id, name, parent_id\n- Inventory: product_id, warehouse_id, quantity, last_updated\n- Warehouses: id, name, location, capacity\n- Transactions: id, product_id, quantity, type, timestamp, user_id"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "What's the deployment process for the mobile application?"}, {"from": "assistant", "value": "1. Merge code to release branch\n2. Run automated test suite\n3. Build the application packages\n4. Submit to internal testing (TestFlight/Firebase)\n5. Verify with QA team\n6. Deploy to production stores (App Store/Google Play)\n7. Monitor crash reports and analytics"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "Show me the API endpoints for the user service"}, {"from": "assistant", "value": "GET /api/users - List all users\nGET /api/users/{id} - Get specific user\nPOST /api/users - Create new user\nPUT /api/users/{id} - Update user\nDELETE /api/users/{id} - Delete user\nPOST /api/users/login - User login\nPOST /api/users/logout - User logout\nGET /api/users/me - Get current user"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "What are the system requirements for running the video processing pipeline?"}, {"from": "assistant", "value": "Minimum requirements:\n- CPU: 8 cores (Intel i7/AMD Ryzen 7 or higher)\n- RAM: 16GB DDR4\n- GPU: NVIDIA with 8GB VRAM (CUDA compatible)\n- Storage: 500GB SSD\n- OS: Ubuntu 22.04 LTS\n- Network: 1Gbps connection"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "What's the current sprint backlog for Team Alpha?"}, {"from": "assistant", "value": "Sprint 14 Backlog (Team Alpha):\n1. Implement OAuth2 authentication - 13 points\n2. Fix pagination bug in search results - 5 points\n3. Add export to CSV functionality - 8 points\n4. Optimize database queries for dashboard - 8 points\n5. Update user documentation - 3 points\n6. Refactor notification service - 13 points"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "What are the configuration options for the logging system?"}, {"from": "assistant", "value": "Logging configuration options:\n\nLOG_LEVEL: [DEBUG, INFO, WARNING, ERROR, CRITICAL]\nLOG_FORMAT: [JSON, TEXT, CSV]\nOUTPUT_DESTINATION: [CONSOLE, FILE, REMOTE]\nROTATION_POLICY: [SIZE, TIME, NONE]\nRETENTION_PERIOD: [1-365 days]\nINCLUDE_METADATA: [true/false]\nADDITIONAL_FIELDS: custom key-value pairs"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "Explain the CI/CD pipeline for the web application"}, {"from": "assistant", "value": "CI/CD Pipeline:\n\n1. Code Push → GitHub repository\n2. Jenkins triggers build job\n3. Static code analysis with SonarQube\n4. Unit tests with Jest\n5. Integration tests with Cypress\n6. Build Docker image\n7. Push to container registry\n8. Deploy to staging environment\n9. Run E2E tests\n10. Deploy to production with Kubernetes"}]} -{"conversations": [{"from": "system", "value": "Project assistant"}, {"from": "user", "value": "What are the performance metrics for the recommendation engine?"}, {"from": "assistant", "value": "Key performance metrics:\n\n- Mean Average Precision (MAP): 0.83\n- Click-Through Rate (CTR): 4.7%\n- Conversion Rate: 2.1%\n- Average Response Time: 120ms\n- p95 Latency: 350ms\n- Daily Active Users (DAU): 125,000\n- Recommendation Diversity Score: 0.68"}]} \ No newline at end of file diff --git a/finetune/offlinetraining.py b/finetune/offlinetraining.py deleted file mode 100644 index 97c7304..0000000 --- a/finetune/offlinetraining.py +++ /dev/null @@ -1,159 +0,0 @@ -import torch -import os -from transformers import AutoModelForCausalLM, AutoTokenizer -from transformers import TrainingArguments -from peft import LoraConfig, get_peft_model -from datasets import load_dataset -from trl import SFTConfig, SFTTrainer - - - - -# Model configuration -max_seq_length = 2048 -# model_name = "meta-llama/Llama-3.2-3B-Instruct" -model_path = r'D:\work\models\Meta-Llama-3.2-3B-Instruct' -# Load model and tokenizer -model = AutoModelForCausalLM.from_pretrained( - model_path, - torch_dtype=torch.bfloat16 if torch.cuda.is_available() and torch.cuda.get_device_capability()[0] >= 8 else torch.float16, - device_map="auto" -) - -tokenizer = AutoTokenizer.from_pretrained( - model_path, - model_max_length=max_seq_length, - padding_side="right" -) - -# Configure LoRA -lora_config = LoraConfig( - r=16, # rank - lora_alpha=16, - target_modules=["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"], - lora_dropout=0, - bias="none", - task_type="CAUSAL_LM" -) - -# Apply PEFT -model = get_peft_model(model, lora_config) - -# Define prompt template for formatting -llama31_prompt = """<|begin_of_text|><|start_header_id|>system<|end_header_id|> - -{}<|eot_id|><|start_header_id|>user<|end_header_id|> - -{}<|eot_id|><|start_header_id|>assistant<|end_header_id|> - -{}<|eot_id|>""" - -def formatting_prompts_func(examples): - fields = examples["conversations"] - texts = [] - for convos in fields: - instruction = convos[0]['value'] - input_text = convos[1]['value'] - output = convos[2]['value'] - text = llama31_prompt.format(instruction, input_text, output) - texts.append(text) - return {"text": texts} - -# Load and process dataset -dataset = load_dataset("json", data_files={"train": "data.jsonl"}, split="train") -dataset = dataset.map(formatting_prompts_func, batched=True) - -# Configure training arguments -training_args = TrainingArguments( - output_dir="outputs", - per_device_train_batch_size=2, - gradient_accumulation_steps=4, - warmup_steps=5, - num_train_epochs=3, - learning_rate=2e-4, - fp16=(torch.cuda.is_available() and not (torch.cuda.get_device_capability()[0] >= 8)), - bf16=(torch.cuda.is_available() and torch.cuda.get_device_capability()[0] >= 8), - logging_steps=1, - optim="adamw_torch", - weight_decay=0.01, - lr_scheduler_type="linear", - seed=3407, - report_to="none" -) - -# Record starting memory usage -if torch.cuda.is_available(): - gpu_stats = torch.cuda.get_device_properties(0) - start_gpu_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3) - max_memory = round(gpu_stats.total_memory / 1024 / 1024 / 1024, 3) - print(f"GPU = {gpu_stats.name}. Max memory = {max_memory} GB.") - print(f"{start_gpu_memory} GB of memory reserved.") - -# Initialize SFT trainer without custom preprocessing for now -sft_config = SFTConfig( - max_seq_length=max_seq_length, - packing=False, - **training_args.to_dict() # Pass training arguments to SFTConfig -) - -# Create trainer using SFTConfig -trainer = SFTTrainer( - model=model, - args=sft_config, - train_dataset=dataset, # Use original dataset - data_collator=None, # Let the trainer handle this -) - -# Train the model -trainer_stats = trainer.train() - -# Report training statistics and memory usage -if torch.cuda.is_available(): - used_memory = round(torch.cuda.max_memory_reserved() / 1024 / 1024 / 1024, 3) - used_memory_for_lora = round(used_memory - start_gpu_memory, 3) - used_percentage = round(used_memory / max_memory * 100, 3) - lora_percentage = round(used_memory_for_lora / max_memory * 100, 3) - print(f"{trainer_stats.metrics['train_runtime']} seconds used for training.") - print(f"{round(trainer_stats.metrics['train_runtime']/60, 2)} minutes used for training.") - print(f"Peak reserved memory = {used_memory} GB.") - print(f"Peak reserved memory for training = {used_memory_for_lora} GB.") - print(f"Peak reserved memory % of max memory = {used_percentage} %.") - print(f"Peak reserved memory for training % of max memory = {lora_percentage} %.") - -# Save the model -output_dir = "model" -os.makedirs(output_dir, exist_ok=True) -model.save_pretrained(output_dir) -tokenizer.save_pretrained(output_dir) - -# Example inference -def run_inference(prompt): - # Set model to evaluation mode - model.eval() - - # Format the messages - messages = [{"role": "user", "content": prompt}] - inputs = tokenizer.apply_chat_template( - messages, - tokenize=True, - add_generation_prompt=True, - return_tensors="pt" - ).to(model.device) - - # Generate response - with torch.no_grad(): - outputs = model.generate( - input_ids=inputs, - max_new_tokens=128, - do_sample=True, - temperature=1.5, - top_p=0.9 - ) - - response = tokenizer.decode(outputs[0], skip_special_tokens=True) - return response - -# Example usage (uncomment to test) -# test_prompt = "Describe a tall tower in the capital of France." -# response = run_inference(test_prompt) -# print(response) \ No newline at end of file diff --git a/finetune/requirements.txt b/finetune/requirements.txt deleted file mode 100644 index fc38b72..0000000 --- a/finetune/requirements.txt +++ /dev/null @@ -1,8 +0,0 @@ -torch>=2.1.0 -transformers>=4.39.0 -datasets>=2.18.0 -accelerate>=0.25.0 -tqdm -huggingface_hub -sentencepiece -protobuf<4 \ No newline at end of file diff --git a/graphrag b/graphrag new file mode 160000 index 0000000..fdb7e38 --- /dev/null +++ b/graphrag @@ -0,0 +1 @@ +Subproject commit fdb7e3835badc483f50605db476d8636aeea7156 diff --git a/graphrag-ollama-config/.env b/graphrag-ollama-config/.env new file mode 100644 index 0000000..2a7e3cc --- /dev/null +++ b/graphrag-ollama-config/.env @@ -0,0 +1,122 @@ +# ============================================================================== +# VeritasGraph - OpenAI Compatible API Environment Configuration +# ============================================================================== +# This file provides example environment variables for OpenAI-compatible APIs. +# Copy this file to `.env` and fill in your values. +# +# Supported providers: +# - OpenAI (api.openai.com) +# - Azure OpenAI +# - Groq (api.groq.com) +# - Together AI (api.together.xyz) +# - Anyscale (api.endpoints.anyscale.com) +# - OpenRouter (openrouter.ai/api) +# - LM Studio (localhost:1234) +# - LocalAI (localhost:8080) +# - vLLM (your-vllm-server:8000) +# - text-generation-webui (localhost:5000) +# - Any other OpenAI-compatible endpoint +# ============================================================================== + +# ============================================================================== +# OPENAI (Native) for LLM +# ============================================================================== +# GRAPHRAG_API_KEY=sk-proj-* +# GRAPHRAG_LLM_MODEL=gpt-4-turbo-preview +# GRAPHRAG_LLM_API_BASE=https://api.openai.com/v1 + +# # ============================================================================== +# # OpenAI Embeddings (use text-embedding-3-small for OpenAI) +# # ============================================================================== +# GRAPHRAG_EMBEDDING_MODEL=text-embedding-3-small +# GRAPHRAG_EMBEDDING_API_BASE=https://api.openai.com/v1 +# GRAPHRAG_EMBEDDING_API_KEY=sk-proj-* + +# # ============================================================================== +# # Ollama for Embeddings (uncomment to use local Ollama instead) +# # ============================================================================== +# GRAPHRAG_EMBEDDING_MODEL=nomic-embed-text +# GRAPHRAG_EMBEDDING_API_BASE=http://localhost:11434/v1 +# GRAPHRAG_EMBEDDING_API_KEY=ollama + +# ============================================================================== +# AZURE OPENAI +# ============================================================================== +# GRAPHRAG_API_KEY=your-azure-openai-key +# GRAPHRAG_LLM_MODEL=gpt-4 # This should match your deployment name +# GRAPHRAG_LLM_API_BASE=https://your-resource.openai.azure.com +# GRAPHRAG_EMBEDDING_MODEL=text-embedding-ada-002 +# GRAPHRAG_EMBEDDING_API_BASE=https://your-resource.openai.azure.com +# Note: Set api_version in settings.yaml for Azure + +# ============================================================================== +# GROQ +# ============================================================================== +# GRAPHRAG_API_KEY=gsk_your-groq-api-key +# GRAPHRAG_LLM_MODEL=llama-3.1-70b-versatile +# GRAPHRAG_LLM_API_BASE=https://api.groq.com/openai/v1 +# GRAPHRAG_EMBEDDING_MODEL=nomic-embed-text # Use Ollama or other for embeddings +# GRAPHRAG_EMBEDDING_API_BASE=http://localhost:11434/v1 + +# ============================================================================== +# TOGETHER AI +# ============================================================================== +# GRAPHRAG_API_KEY=your-together-api-key +# GRAPHRAG_LLM_MODEL=meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo +# GRAPHRAG_LLM_API_BASE=https://api.together.xyz/v1 +# GRAPHRAG_EMBEDDING_MODEL=togethercomputer/m2-bert-80M-8k-retrieval +# GRAPHRAG_EMBEDDING_API_BASE=https://api.together.xyz/v1 + +# ============================================================================== +# OPENROUTER +# ============================================================================== +# GRAPHRAG_API_KEY=sk-or-your-openrouter-key +# GRAPHRAG_LLM_MODEL=anthropic/claude-3.5-sonnet +# GRAPHRAG_LLM_API_BASE=https://openrouter.ai/api/v1 +# GRAPHRAG_EMBEDDING_MODEL=openai/text-embedding-3-small +# GRAPHRAG_EMBEDDING_API_BASE=https://openrouter.ai/api/v1 + +# ============================================================================== +# LM STUDIO (Local) +# ============================================================================== +# GRAPHRAG_API_KEY=lm-studio # LM Studio doesn't require a real key +# GRAPHRAG_LLM_MODEL=local-model # Use the model name loaded in LM Studio +# GRAPHRAG_LLM_API_BASE=http://localhost:1234/v1 +# GRAPHRAG_EMBEDDING_MODEL=nomic-embed-text # Use Ollama for embeddings +# GRAPHRAG_EMBEDDING_API_BASE=http://localhost:11434/v1 + +# ============================================================================== +# VLLM Server +# ============================================================================== +# GRAPHRAG_API_KEY=vllm # vLLM may not require authentication +# GRAPHRAG_LLM_MODEL=meta-llama/Meta-Llama-3.1-8B-Instruct +# GRAPHRAG_LLM_API_BASE=http://localhost:8000/v1 +# GRAPHRAG_EMBEDDING_MODEL=BAAI/bge-base-en-v1.5 +# GRAPHRAG_EMBEDDING_API_BASE=http://localhost:8001/v1 + +# ============================================================================== +# LOCALAI +# ============================================================================== +# GRAPHRAG_API_KEY=localai # LocalAI doesn't require a real key by default +# GRAPHRAG_LLM_MODEL=gpt-4 # Or whatever model name you've configured +# GRAPHRAG_LLM_API_BASE=http://localhost:8080/v1 +# GRAPHRAG_EMBEDDING_MODEL=text-embedding-ada-002 +# GRAPHRAG_EMBEDDING_API_BASE=http://localhost:8080/v1 + +# ============================================================================== +# ANYSCALE ENDPOINTS +# ============================================================================== +# GRAPHRAG_API_KEY=your-anyscale-api-key +# GRAPHRAG_LLM_MODEL=meta-llama/Meta-Llama-3.1-70B-Instruct +# GRAPHRAG_LLM_API_BASE=https://api.endpoints.anyscale.com/v1 +# GRAPHRAG_EMBEDDING_MODEL=thenlper/gte-large +# GRAPHRAG_EMBEDDING_API_BASE=https://api.endpoints.anyscale.com/v1 + +# ============================================================================== +# OLLAMA (Default - included for reference) +# ============================================================================== +GRAPHRAG_API_KEY=ollama +GRAPHRAG_LLM_MODEL=qwen3:latest +GRAPHRAG_LLM_API_BASE=http://localhost:11434/v1 +GRAPHRAG_EMBEDDING_MODEL=nomic-embed-text:latest +GRAPHRAG_EMBEDDING_API_BASE=http://localhost:11434/v1 diff --git a/graphrag-ollama-config/app.py b/graphrag-ollama-config/app.py new file mode 100644 index 0000000..6cd1e2b --- /dev/null +++ b/graphrag-ollama-config/app.py @@ -0,0 +1,892 @@ +import gradio as gr +import os +import asyncio +import pandas as pd +import tiktoken +from dotenv import load_dotenv + +# Graph visualization +from graph_visualizer import ( + create_graph_html_for_query, + get_graph_stats, + load_graph_data, + extract_entities_from_response +) + +from graphrag.query.indexer_adapters import read_indexer_entities, read_indexer_reports +from graphrag.query.structured_search.global_search.community_context import GlobalCommunityContext +from graphrag.query.structured_search.global_search.search import GlobalSearch +from graphrag.query.llm.oai.chat_openai import ChatOpenAI +from graphrag.query.question_gen.local_gen import LocalQuestionGen +from graphrag.query.context_builder.entity_extraction import EntityVectorStoreKey +from graphrag.query.indexer_adapters import ( + read_indexer_covariates, + read_indexer_entities, + read_indexer_relationships, + read_indexer_reports, + read_indexer_text_units, +) +from graphrag.query.input.loaders.dfs import ( + store_entity_semantic_embeddings, +) +from graphrag.query.llm.oai.embedding import OpenAIEmbedding +from graphrag.query.question_gen.local_gen import LocalQuestionGen +from graphrag.query.structured_search.local_search.mixed_context import ( + LocalSearchMixedContext, +) +from graphrag.query.structured_search.local_search.search import LocalSearch +from graphrag.vector_stores.lancedb import LanceDBVectorStore + +# Import OpenAI-compatible API configuration from separate module +from openai_config import get_api_type, get_llm_config, get_embedding_config + +# Import Instant Knowledge ingest module +from ingest import ( + ingest_url, + ingest_text_content, + trigger_graphrag_index, + trigger_graphrag_index_async, + trigger_graphrag_index_with_progress, + get_indexing_status, + list_input_files, + delete_input_file, + get_file_preview, + check_dependencies +) + +# Import PageIndex-inspired Reasoning Search (for 99%+ accuracy) +try: + from reasoning_search import ( + enhanced_search, + reasoning_local_search, + reasoning_global_search, + hybrid_reasoning_search, + ReasoningSearchEngine + ) + REASONING_SEARCH_AVAILABLE = True +except ImportError: + REASONING_SEARCH_AVAILABLE = False + print("Warning: reasoning_search module not available. Using standard search.") + +# Load .env from the same directory as this script +script_dir = os.path.dirname(os.path.abspath(__file__)) +load_dotenv(os.path.join(script_dir, '.env')) +join = os.path.join + +PRESET_MAPPING = { + "Default": { + "community_level": 2, + "response_type": "Multiple Paragraphs" + }, + "Detailed": { + "community_level": 4, + "response_type": "Multi-Page Report" + }, + "Quick": { + "community_level": 1, + "response_type": "Single Paragraph" + }, + "Bullet": { + "community_level": 2, + "response_type": "List of 3-7 Points" + }, + "Comprehensive": { + "community_level": 5, + "response_type": "Multi-Page Report" + }, + "High-Level": { + "community_level": 1, + "response_type": "Single Page" + }, + "Focused": { + "community_level": 3, + "response_type": "Multiple Paragraphs" + } +} + +# Query type options with reasoning-based search +QUERY_TYPE_OPTIONS = [ + "global", # Standard community-based search + "local", # Standard entity-based search + "reasoning", # PageIndex-inspired reasoning search (highest accuracy) + "hybrid", # Combined local + global with reasoning +] + +async def global_search(query, input_dir, community_level=2, temperature=0.5, response_type="Multiple Paragraphs"): + llm_config = get_llm_config() + + llm = ChatOpenAI( + api_key=llm_config["api_key"], + api_base=llm_config["api_base"], + model=llm_config["model"], + api_type=llm_config["api_type"], + max_retries=llm_config["max_retries"], + ) + + token_encoder = tiktoken.get_encoding("cl100k_base") + + COMMUNITY_REPORT_TABLE = "create_final_community_reports" + ENTITY_TABLE = "create_final_nodes" + ENTITY_EMBEDDING_TABLE = "create_final_entities" + + entity_df = pd.read_parquet(join(input_dir, f"{ENTITY_TABLE}.parquet")) + report_df = pd.read_parquet(join(input_dir, f"{COMMUNITY_REPORT_TABLE}.parquet")) + entity_embedding_df = pd.read_parquet(join(input_dir, f"{ENTITY_EMBEDDING_TABLE}.parquet")) + + reports = read_indexer_reports(report_df, entity_df, community_level) + entities = read_indexer_entities(entity_df, entity_embedding_df, community_level) + + context_builder = GlobalCommunityContext( + community_reports=reports, + entities=entities, + token_encoder=token_encoder, + ) + + context_builder_params = { + "use_community_summary": False, # False means using full community reports. True means using community short summaries. + "shuffle_data": True, + "include_community_rank": True, + "min_community_rank": 0, + "community_rank_name": "rank", + "include_community_weight": True, + "community_weight_name": "occurrence weight", + "normalize_community_weight": True, + "max_tokens": 4000, # change this based on the token limit you have on your model (if you are using a model with 8k limit, a good setting could be 5000) + "context_name": "Reports", + } + + map_llm_params = { + "max_tokens": 1000, + "temperature": temperature, + "response_format": {"type": "json_object"}, + } + + reduce_llm_params = { + "max_tokens": 2000, # change this based on the token limit you have on your model (if you are using a model with 8k limit, a good setting could be 1000-1500) + "temperature": temperature, + } + + search_engine = GlobalSearch( + llm=llm, + context_builder=context_builder, + token_encoder=token_encoder, + max_data_tokens=5000, # change this based on the token limit you have on your model (if you are using a model with 8k limit, a good setting could be 5000) + map_llm_params=map_llm_params, + reduce_llm_params=reduce_llm_params, + allow_general_knowledge=False, # set this to True will add instruction to encourage the LLM to incorporate general knowledge in the response, which may increase hallucinations, but could be useful in some use cases. + json_mode=True, # set this to False if your LLM model does not support JSON mode. + context_builder_params=context_builder_params, + concurrent_coroutines=1, + response_type=response_type, # free form text describing the response type and format, can be anything, e.g. prioritized list, single paragraph, multiple paragraphs, multiple-page report + ) + + result = await search_engine.asearch(query) + return result.response + +def prepare_local_search(input_dir, community_level=2, temperature=0.5): + LANCEDB_URI = f"{input_dir}/lancedb" + + COMMUNITY_REPORT_TABLE = "create_final_community_reports" + ENTITY_TABLE = "create_final_nodes" + ENTITY_EMBEDDING_TABLE = "create_final_entities" + RELATIONSHIP_TABLE = "create_final_relationships" + COVARIATE_TABLE = "create_final_covariates" + TEXT_UNIT_TABLE = "create_final_text_units" + + entity_df = pd.read_parquet(join(input_dir, f"{ENTITY_TABLE}.parquet")) + entity_embedding_df = pd.read_parquet(join(input_dir, f"{ENTITY_EMBEDDING_TABLE}.parquet")) + + entities = read_indexer_entities(entity_df, entity_embedding_df, community_level) + + # load description embeddings to an in-memory lancedb vectorstore + # to connect to a remote db, specify url and port values. + description_embedding_store = LanceDBVectorStore( + collection_name="entity_description_embeddings", + ) + description_embedding_store.connect(db_uri=LANCEDB_URI) + entity_description_embeddings = store_entity_semantic_embeddings( + entities=entities, vectorstore=description_embedding_store + ) + + relationship_df = pd.read_parquet(join(input_dir, f"{RELATIONSHIP_TABLE}.parquet")) + relationships = read_indexer_relationships(relationship_df) + + # covariate_df = pd.read_parquet(join(input_dir, f"{COVARIATE_TABLE}.parquet")) + # claims = read_indexer_covariates(covariate_df) + # covariates = {"claims": claims} + + report_df = pd.read_parquet(join(input_dir, f"{COMMUNITY_REPORT_TABLE}.parquet")) + reports = read_indexer_reports(report_df, entity_df, community_level) + + text_unit_df = pd.read_parquet(join(input_dir, f"{TEXT_UNIT_TABLE}.parquet")) + text_units = read_indexer_text_units(text_unit_df) + + llm_config = get_llm_config() + embedding_config = get_embedding_config() + + llm = ChatOpenAI( + api_key=llm_config["api_key"], + api_base=llm_config["api_base"], + model=llm_config["model"], + api_type=llm_config["api_type"], + max_retries=llm_config["max_retries"], + ) + + token_encoder = tiktoken.get_encoding("cl100k_base") + + text_embedder = OpenAIEmbedding( + api_key=embedding_config["api_key"], + api_base=embedding_config["api_base"], + api_type=embedding_config["api_type"], + model=embedding_config["model"], + deployment_name=embedding_config["deployment_name"], + max_retries=embedding_config["max_retries"], + ) + + context_builder = LocalSearchMixedContext( + community_reports=reports, + text_units=text_units, + entities=entities, + relationships=relationships, + entity_text_embeddings=description_embedding_store, + embedding_vectorstore_key=EntityVectorStoreKey.ID, # if the vectorstore uses entity title as ids, set this to EntityVectorStoreKey.TITLE + text_embedder=text_embedder, + token_encoder=token_encoder, + ) + + local_context_params = { + "text_unit_prop": 0.5, + "community_prop": 0.1, + "conversation_history_max_turns": 5, + "conversation_history_user_turns_only": True, + "top_k_mapped_entities": 10, + "top_k_relationships": 10, + "include_entity_rank": True, + "include_relationship_weight": True, + "include_community_rank": False, + "return_candidate_context": False, + "embedding_vectorstore_key": EntityVectorStoreKey.ID, # set this to EntityVectorStoreKey.TITLE if the vectorstore uses entity title as ids + "max_tokens": 5000, # change this based on the token limit you have on your model (if you are using a model with 8k limit, a good setting could be 5000) + } + + llm_params = { + "max_tokens": 1500, # change this based on the token limit you have on your model (if you are using a model with 8k limit, a good setting could be 1000=1500) + "temperature": temperature, + } + + return llm, context_builder, token_encoder, llm_params, local_context_params + +async def local_search(query, input_dir, community_level=2, temperature=0.5, response_type="Multiple Paragraphs"): + ( + llm, + context_builder, + token_encoder, + llm_params, + local_context_params + ) = prepare_local_search(input_dir, community_level, temperature) + + search_engine = LocalSearch( + llm=llm, + context_builder=context_builder, + token_encoder=token_encoder, + llm_params=llm_params, + context_builder_params=local_context_params, + response_type=response_type, # free form text describing the response type and format, can be anything, e.g. prioritized list, single paragraph, multiple paragraphs, multiple-page report + ) + + result = await search_engine.asearch(query) + return result.response + +async def local_question_generate(question_history, input_dir, community_level=2, temperature=0.5): + ( + llm, + context_builder, + token_encoder, + llm_params, + local_context_params + ) = prepare_local_search(input_dir, community_level, temperature) + + question_generator = LocalQuestionGen( + llm=llm, + context_builder=context_builder, + token_encoder=token_encoder, + llm_params=llm_params, + context_builder_params=local_context_params, + ) + + # Ensure question_history is a list of strings (not nested lists) + # If empty, provide a default starting question + if not question_history: + question_history = ["What are the main topics in this dataset?"] + + # Flatten any nested lists and ensure all items are strings + flat_history = [] + for item in question_history: + if isinstance(item, list): + flat_history.extend([str(x) for x in item]) + else: + flat_history.append(str(item)) + + result = await question_generator.agenerate( + question_history=flat_history, context_data=None, question_count=5 + ) + return result.response + + +async def chat_graphrag( + query, + history, + selected_folder, + query_type, + temperature, + preset, + show_graph=True + ): + # Handle both new format ("output" -> output/artifacts) and old format (timestamp -> output/timestamp/artifacts) + if selected_folder == "output": + input_dir = join(script_dir, "output", "artifacts") + else: + input_dir = join(script_dir, "output", selected_folder, "artifacts") + + community_level = PRESET_MAPPING[preset]["community_level"] + response_type = PRESET_MAPPING[preset]["response_type"] + + response = None + query_entities = [] + + if query == "/generate": + # Extract user messages from history (messages format: list of dicts with 'role' and 'content') + question_history = [msg["content"] for msg in history if msg.get("role") == "user"] + response = await local_question_generate( + question_history, input_dir, community_level, temperature + ) + elif query_type == "reasoning" and REASONING_SEARCH_AVAILABLE: + # PageIndex-inspired reasoning-based search (highest accuracy) + result = await enhanced_search( + query=query, + input_dir=input_dir, + query_type="auto", + community_level=community_level, + temperature=temperature, + response_type=response_type + ) + response = result["response"] + # Add confidence and verification info + confidence = result.get("confidence", 0) + verified = result.get("verified", False) + if confidence > 0: + response += f"\n\n---\n📊 **Confidence:** {confidence:.0%} | ✅ **Verified:** {verified}" + query_entities = [word for word in query.split() if len(word) > 3] + elif query_type == "hybrid" and REASONING_SEARCH_AVAILABLE: + # Hybrid reasoning search (combined local + global) + result = await hybrid_reasoning_search( + query=query, + input_dir=input_dir, + community_level=community_level, + temperature=temperature, + response_type=response_type + ) + response = result["response"] + confidence = result.get("confidence", 0) + if confidence > 0: + response += f"\n\n---\n📊 **Confidence:** {confidence:.0%} | 🔄 **Strategy:** Hybrid" + query_entities = [word for word in query.split() if len(word) > 3] + elif query_type == "global": + response = await global_search( + query, input_dir, community_level, temperature, response_type + ) + # Extract key terms from query for graph visualization + query_entities = [word for word in query.split() if len(word) > 3] + elif query_type == "local": + response = await local_search( + query, input_dir, community_level, temperature, response_type + ) + query_entities = [word for word in query.split() if len(word) > 3] + else: + response = "Sorry, I can't do a search for you right now" + + print(response) + history.append({"role": "user", "content": query}) + history.append({"role": "assistant", "content": response}) + + # Generate graph visualization if enabled + graph_html = "" + if show_graph and query != "/generate": + try: + # Try to extract mentioned entities from response + entity_df, _, _ = load_graph_data(input_dir) + response_entities = extract_entities_from_response(response, entity_df) + all_entities = list(set(query_entities + response_entities)) + + graph_html = create_graph_html_for_query( + input_dir, + query_entities=all_entities[:10], + max_nodes=40 + ) + except Exception as e: + graph_html = f"
Graph visualization unavailable: {str(e)}
" + + return "", history, graph_html + +def list_output_folders(): + """List available output folders for GraphRAG queries. + + Supports both old format (timestamped folders like 20241201-123456) + and new format (direct artifacts/ folder in output/). + """ + output_dir = join(script_dir, "output") + if not os.path.exists(output_dir): + return [] + + # Check for new GraphRAG format (artifacts directly in output/) + if os.path.exists(join(output_dir, "artifacts")): + return ["output"] # Return "output" as the folder choice + + # Check for old format (timestamped folders) + folders = [f for f in os.listdir(output_dir) if os.path.isdir(join(output_dir, f)) and f[0].isdigit()] + return sorted(folders, reverse=True) + +def create_gradio_interface(): + # Sample prompts for developers to try (based on Student Visa & Athlete Recruitment data) + SAMPLE_PROMPTS = [ + "What are the main eligibility criteria for student visas across different countries?", + "Compare the visa requirements between USA (F-1) and UK (Tier 4) student visas", + "What are the top reasons candidates are rejected for student visas?", + "How do NCAA eligibility requirements relate to academic performance?", + "What financial requirements exist for different visa types?", + "/generate", # Generate follow-up questions + ] + + custom_css = """ + .contain { display: flex; flex-direction: column; } + + #component-0 { height: 100%; } + + #main-container { display: flex; height: 100%; } + + #right-column { height: calc(100vh - 100px); } + + #chatbot { flex-grow: 1; overflow: auto; } + + .sample-prompts { margin-top: 10px; } + .sample-prompts button { margin: 2px; font-size: 12px; } + + """ + with gr.Blocks(css=custom_css, theme=gr.themes.Base(), title="VeritasGraph - GraphRAG Demo") as demo: + gr.Markdown(""" + # 🔍 VeritasGraph - Graph RAG Demo + **Enterprise-Grade Knowledge Graph RAG with Verifiable Attribution** + + 📊 **Dataset:** Student Visa & Admission Eligibility + Elite Athlete Recruitment Analytics + + Try the sample prompts below or enter your own question! + """) + + with gr.Row(elem_id="main-container"): + with gr.Column(scale=1, elem_id="left-column"): + output_folders = list_output_folders() + output_folder = output_folders[0] if output_folders else "No output found" + selected_folder = gr.Dropdown( + label="Select Output Folder", + choices=output_folders if output_folders else ["No output found"], + value=output_folder, + interactive=True, + allow_custom_value=True + ) + + query_type = gr.Radio( + ["global", "local", "reasoning", "hybrid"], + label="Query Type", + value="reasoning" if REASONING_SEARCH_AVAILABLE else "global", + info="🎯 Reasoning: PageIndex-style (99% acc) | 🔄 Hybrid: Combined | 🌐 Global: Communities | 📍 Local: Entities" + ) + + temperature = gr.Slider( + label="Temperature", + minimum=0.0, + maximum=2.0, + step=0.1, + value=float(0.5) + ) + + preset = gr.Radio( + ["Default", "Detailed", "Quick", "Bullet", "Comprehensive", "High-Level", "Focused"], + label="Preset", + value="Default", + info="How specified is the query result" + ) + + # Graph visualization toggle + show_graph = gr.Checkbox( + label="🔗 Show Graph Visualization", + value=True, + info="Display interactive knowledge graph after each query" + ) + + with gr.Column(scale=2, elem_id="right-column"): + with gr.Tabs(): + with gr.Tab("💬 Chat", id="chat-tab"): + chatbot = gr.Chatbot( + label="Chat History", + elem_id="chatbot", + height=400, + value=[{"role": "assistant", "content": """👋 Welcome to **VeritasGraph**! + +I can help you explore the **Student Visa & Athlete Recruitment** knowledge graph. + +**📊 Dataset includes:** +- 7,773 student visa candidates (USA F-1, UK Tier 4, Schengen) +- 5,432 elite athletes (FIFA, NCAA, UK GBE compliance) + +**Try these sample prompts:** +- "What are the eligibility criteria for student visas?" +- "Compare USA vs UK visa requirements" +- "What causes visa rejections?" +- "How does NCAA eligibility work?" + +**Tips:** +- 🎯 **Reasoning Search** (NEW!) - PageIndex-inspired, 99% accuracy with verification +- 🔄 **Hybrid Search** - Combines local + global with LLM reasoning +- 🌐 **Global Search** - High-level summaries across all data +- 📍 **Local Search** - Specific entity queries (e.g., "NCAA", "F-1 visa") +- Type `/generate` to get AI-suggested follow-up questions +- Toggle **Show Graph Visualization** to see the knowledge graph! + +What would you like to know?"""}] + ) + + with gr.Tab("🔗 Graph Explorer", id="graph-tab"): + gr.Markdown(""" + ### Interactive Knowledge Graph + The graph updates automatically after each query, showing entities and relationships used in the response. + + **Legend:** + - 🔴 **Red nodes** = Query-related entities + - 🔵 **Colored nodes** = Communities (groups of related entities) + - **Node size** = Importance (connection count) + - **Hover** for entity details | **Drag** to rearrange | **Scroll** to zoom + """) + graph_display = gr.HTML( + value="

🔗 Knowledge Graph

Run a query to see the related subgraph visualization

", + elem_id="graph-display" + ) + + with gr.Tab("⚡ Instant Knowledge", id="ingest-tab"): + # Check dependencies + deps = check_dependencies() + dep_status = [] + if deps['youtube']: + dep_status.append("✅ YouTube") + else: + dep_status.append("❌ YouTube (install: pip install youtube-transcript-api yt-dlp)") + if deps['web']: + dep_status.append("✅ Web Articles") + else: + dep_status.append("❌ Web Articles (install: pip install trafilatura)") + + gr.Markdown(f""" + ### ⚡ Instant Knowledge Ingest + **Paste a YouTube URL or Web Article URL** to instantly add it to your knowledge graph! + + **Supported Sources:** + - 📺 **YouTube Videos** - Automatically extracts transcripts (including auto-generated captions) + - 📰 **Web Articles** - Extracts main content from blog posts, news articles, documentation + + **Status:** {' | '.join(dep_status)} + """) + + with gr.Row(): + ingest_url_input = gr.Textbox( + label="🔗 Source URL", + placeholder="Paste YouTube or article URL here... (e.g., https://youtube.com/watch?v=... or https://blog.example.com/article)", + scale=4 + ) + ingest_btn = gr.Button("⚡ Ingest URL", variant="primary", scale=1) + + gr.Markdown("---") + gr.Markdown(""" + ### 📝 Paste Text Content + **Copy-paste text content** directly from files, documents, PDFs, or any other source! + """) + + with gr.Row(): + text_title_input = gr.Textbox( + label="📌 Title", + placeholder="Enter a descriptive title for this content...", + scale=2 + ) + + text_content_input = gr.Textbox( + label="📄 Text Content", + placeholder="Paste your text content here... (minimum 50 characters)", + lines=8, + max_lines=20 + ) + + with gr.Row(): + text_ingest_btn = gr.Button("📝 Add Text to Knowledge", variant="primary") + text_clear_btn = gr.Button("🗑️ Clear", variant="secondary") + + ingest_status = gr.Markdown( + value="*Paste a URL or text content above and click the corresponding button to add to your knowledge base.*" + ) + + with gr.Row(): + index_btn = gr.Button("📊 Full Index", variant="primary") + update_index_btn = gr.Button("🔄 Update Index", variant="secondary") + check_status_btn = gr.Button("📋 Check Status", variant="secondary") + refresh_files_btn = gr.Button("🔃 Refresh Files", variant="secondary") + + gr.Markdown("---") + gr.Markdown("### 📁 Input Files") + + # File list table + def format_file_list(): + files = list_input_files() + if not files: + return "*No files in input folder yet. Ingest some content above!*" + rows = ["| File | Size | Modified |\n|------|------|----------|"] + for f in files[:20]: # Limit to 20 most recent + size_kb = f['size'] / 1024 + rows.append(f"| {f['name']} | {size_kb:.1f} KB | {f['modified']} |") + if len(files) > 20: + rows.append(f"\n*... and {len(files) - 20} more files*") + return '\n'.join(rows) + + file_list_display = gr.Markdown(value=format_file_list()) + + with gr.Accordion("📄 File Preview", open=False): + file_select = gr.Dropdown( + label="Select file to preview", + choices=[f['name'] for f in list_input_files()], + interactive=True + ) + file_preview = gr.Textbox( + label="Content Preview", + lines=10, + max_lines=20, + interactive=False + ) + delete_file_btn = gr.Button("🗑️ Delete Selected File", variant="stop") + + # Ingest button handler + def handle_ingest(url): + if not url or not url.strip(): + return "⚠️ Please enter a URL first." + success, message, filepath = ingest_url(url) + return message + + ingest_btn.click( + fn=handle_ingest, + inputs=[ingest_url_input], + outputs=[ingest_status] + ).then( + fn=format_file_list, + outputs=[file_list_display] + ).then( + fn=lambda: gr.update(choices=[f['name'] for f in list_input_files()]), + outputs=[file_select] + ) + + # Text content ingest button handler + def handle_text_ingest(title, content): + success, message, filepath = ingest_text_content(title, content) + return message + + text_ingest_btn.click( + fn=handle_text_ingest, + inputs=[text_title_input, text_content_input], + outputs=[ingest_status] + ).then( + fn=format_file_list, + outputs=[file_list_display] + ).then( + fn=lambda: gr.update(choices=[f['name'] for f in list_input_files()]), + outputs=[file_select] + ).then( + fn=lambda: ("", ""), + outputs=[text_title_input, text_content_input] + ) + + # Clear text content button handler + text_clear_btn.click( + fn=lambda: ("", ""), + outputs=[text_title_input, text_content_input] + ) + + # Full Index button handler - runs full index with streaming progress + def handle_full_index(): + for progress_msg in trigger_graphrag_index_with_progress(update_mode=False): + yield progress_msg + + index_btn.click( + fn=handle_full_index, + outputs=[ingest_status] + ) + + # Update Index button handler - runs incremental update with streaming progress + def handle_update_index(): + for progress_msg in trigger_graphrag_index_with_progress(update_mode=True): + yield progress_msg + + update_index_btn.click( + fn=handle_update_index, + outputs=[ingest_status] + ) + + # Check status button handler + def handle_check_status(): + status, is_complete = get_indexing_status() + return status + + check_status_btn.click( + fn=handle_check_status, + outputs=[ingest_status] + ) + + # Refresh file list + refresh_files_btn.click( + fn=format_file_list, + outputs=[file_list_display] + ).then( + fn=lambda: gr.update(choices=[f['name'] for f in list_input_files()]), + outputs=[file_select] + ) + + # File preview handler + def handle_preview(filename): + if filename: + return get_file_preview(filename, max_chars=2000) + return "" + + file_select.change( + fn=handle_preview, + inputs=[file_select], + outputs=[file_preview] + ) + + # Delete file handler + def handle_delete(filename): + if filename: + success, msg = delete_input_file(filename) + return msg, format_file_list(), gr.update(choices=[f['name'] for f in list_input_files()], value=None), "" + return "⚠️ No file selected", format_file_list(), gr.update(choices=[f['name'] for f in list_input_files()]), "" + + delete_file_btn.click( + fn=handle_delete, + inputs=[file_select], + outputs=[ingest_status, file_list_display, file_select, file_preview] + ) + + # Sample prompts as clickable examples + gr.Markdown("**📝 Sample Prompts (click to use):**", elem_classes=["sample-prompts"]) + with gr.Row(): + example_btns = [] + example_prompts = [ + ("🎓 Visa Criteria", "What are the main eligibility criteria for student visas across different countries?"), + ("🆚 USA vs UK", "Compare the visa requirements between USA F-1 and UK Tier 4 student visas"), + ("⚽ NCAA Rules", "How do NCAA eligibility requirements relate to academic and athletic performance?"), + ("💡 Generate", "/generate"), + ] + + with gr.Row(): + for label, prompt in example_prompts[:2]: + btn = gr.Button(label, size="sm", variant="secondary") + example_btns.append((btn, prompt)) + with gr.Row(): + for label, prompt in example_prompts[2:]: + btn = gr.Button(label, size="sm", variant="secondary") + example_btns.append((btn, prompt)) + + with gr.Row(): + query = gr.Textbox( + label="Input", + placeholder="Enter your query here or click a sample prompt above...", + elem_id="query-input", + scale=3 + ) + query_btn = gr.Button("Send Query", variant="primary") + + # Connect example buttons to fill in the query + for btn, prompt in example_btns: + btn.click(lambda p=prompt: p, outputs=[query]) + + # Query submission with graph visualization + query.submit( + fn=chat_graphrag, + inputs=[ + query, + chatbot, + selected_folder, + query_type, + temperature, + preset, + show_graph + ], + outputs=[query, chatbot, graph_display] + ) + query_btn.click( + fn=chat_graphrag, + inputs=[ + query, + chatbot, + selected_folder, + query_type, + temperature, + preset, + show_graph + ], + outputs=[query, chatbot, graph_display] + ) + + # Standalone graph explorer function + def explore_full_graph(selected_folder, max_nodes=50): + """Show the full knowledge graph (top nodes by connectivity).""" + if selected_folder == "output": + input_dir = join("output", "artifacts") + else: + input_dir = join("output", selected_folder, "artifacts") + + return create_graph_html_for_query(input_dir, query_entities=[], max_nodes=max_nodes) + + # Add a button to explore full graph + with gr.Row(): + explore_btn = gr.Button("🔍 Explore Full Graph", variant="secondary", size="sm") + explore_btn.click( + fn=explore_full_graph, + inputs=[selected_folder], + outputs=[graph_display] + ) + + return demo.queue() + + +demo = create_gradio_interface() +app = demo.app + +# Path to graph cache directory for file serving +GRAPH_CACHE_DIR = os.path.join(os.path.dirname(__file__), "graph_cache") +os.makedirs(GRAPH_CACHE_DIR, exist_ok=True) + +if __name__ == "__main__": + import argparse + parser = argparse.ArgumentParser(description="VeritasGraph - GraphRAG Demo") + parser.add_argument("--share", action="store_true", help="Create a public shareable link") + parser.add_argument("--port", type=int, default=7860, help="Port to run the server on") + parser.add_argument("--host", type=str, default="127.0.0.1", help="Host to bind to (use 0.0.0.0 for external access)") + args = parser.parse_args() + + print("\n" + "="*60) + print("🚀 VeritasGraph - GraphRAG Demo Server") + print("="*60) + if args.share: + print("📡 Creating public shareable link...") + print(f"🌐 Local URL: http://{args.host}:{args.port}") + print("="*60 + "\n") + + demo.launch( + server_port=args.port, + server_name=args.host, + share=args.share, + allowed_paths=[GRAPH_CACHE_DIR] + ) \ No newline at end of file diff --git a/graphrag-ollama-config/graph_visualizer.py b/graphrag-ollama-config/graph_visualizer.py new file mode 100644 index 0000000..8add478 --- /dev/null +++ b/graphrag-ollama-config/graph_visualizer.py @@ -0,0 +1,355 @@ +""" +Graph Visualization Module for VeritasGraph +Creates interactive 2D/3D graph visualizations using PyVis +""" + +import os +import pandas as pd +import networkx as nx +from pyvis.network import Network +import json +import base64 +from typing import Optional, List, Dict, Any, Tuple + +join = os.path.join + + +def load_graph_data(input_dir: str) -> Tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]: + """Load entities, relationships, and communities from parquet files.""" + + ENTITY_TABLE = "create_final_nodes" + RELATIONSHIP_TABLE = "create_final_relationships" + COMMUNITY_TABLE = "create_final_communities" + + entity_df = pd.read_parquet(join(input_dir, f"{ENTITY_TABLE}.parquet")) + relationship_df = pd.read_parquet(join(input_dir, f"{RELATIONSHIP_TABLE}.parquet")) + + try: + community_df = pd.read_parquet(join(input_dir, f"{COMMUNITY_TABLE}.parquet")) + except: + community_df = pd.DataFrame() + + return entity_df, relationship_df, community_df + + +def create_full_graph(input_dir: str) -> nx.Graph: + """Create a NetworkX graph from the indexed data.""" + + entity_df, relationship_df, _ = load_graph_data(input_dir) + + G = nx.Graph() + + # Add nodes + for _, row in entity_df.iterrows(): + title = row.get('title', row.get('name', 'Unknown')) + G.add_node( + title, + title=title, + type=row.get('type', 'entity'), + description=row.get('description', ''), + community=row.get('community', 0), + degree=row.get('degree', 1), + ) + + # Add edges + for _, row in relationship_df.iterrows(): + source = row.get('source', '') + target = row.get('target', '') + if source and target and source in G.nodes and target in G.nodes: + G.add_edge( + source, + target, + description=row.get('description', ''), + weight=row.get('weight', 1), + rank=row.get('rank', 1), + ) + + return G + + +def extract_subgraph_for_query( + G: nx.Graph, + query_entities: List[str], + max_depth: int = 2, + max_nodes: int = 50 +) -> nx.Graph: + """Extract a subgraph centered around query-relevant entities.""" + + if not query_entities: + # Return top nodes by degree if no specific entities + top_nodes = sorted(G.nodes(), key=lambda x: G.degree(x), reverse=True)[:max_nodes] + return G.subgraph(top_nodes).copy() + + # Find matching nodes (case-insensitive partial match) + matched_nodes = set() + for entity in query_entities: + entity_lower = entity.lower() + for node in G.nodes(): + if entity_lower in node.lower(): + matched_nodes.add(node) + + if not matched_nodes: + # Fallback to top nodes + top_nodes = sorted(G.nodes(), key=lambda x: G.degree(x), reverse=True)[:max_nodes] + return G.subgraph(top_nodes).copy() + + # Expand to neighbors within max_depth + expanded_nodes = set(matched_nodes) + current_frontier = matched_nodes + + for _ in range(max_depth): + next_frontier = set() + for node in current_frontier: + neighbors = set(G.neighbors(node)) + next_frontier.update(neighbors - expanded_nodes) + expanded_nodes.update(next_frontier) + current_frontier = next_frontier + + if len(expanded_nodes) >= max_nodes: + break + + # Limit to max_nodes, prioritizing query entities and high-degree nodes + if len(expanded_nodes) > max_nodes: + # Sort by: query entity first, then degree + sorted_nodes = sorted( + expanded_nodes, + key=lambda x: (x not in matched_nodes, -G.degree(x)) + ) + expanded_nodes = set(sorted_nodes[:max_nodes]) + + return G.subgraph(expanded_nodes).copy() + + +def get_node_color(node_type: str, is_query_entity: bool = False) -> str: + """Get color based on node type.""" + if is_query_entity: + return "#ff6b6b" # Red for query entities + + color_map = { + "person": "#4ecdc4", + "organization": "#45b7d1", + "location": "#96ceb4", + "event": "#ffeaa7", + "concept": "#dfe6e9", + "document": "#a29bfe", + "default": "#74b9ff", + } + + return color_map.get(node_type.lower(), color_map["default"]) + + +def get_community_color(community_id: int) -> str: + """Get color based on community ID.""" + colors = [ + "#e74c3c", "#3498db", "#2ecc71", "#9b59b6", "#f39c12", + "#1abc9c", "#e91e63", "#00bcd4", "#ff9800", "#8bc34a", + "#673ab7", "#009688", "#ff5722", "#607d8b", "#795548" + ] + return colors[community_id % len(colors)] + + +def create_pyvis_graph( + subgraph: nx.Graph, + query_entities: List[str] = None, + height: str = "600px", + width: str = "100%", + bgcolor: str = "#0a0a0a", + font_color: str = "white", + color_by: str = "community" # "community" or "type" +) -> str: + """Create an interactive PyVis visualization.""" + + query_entities = query_entities or [] + query_entities_lower = [e.lower() for e in query_entities] + + # Create PyVis network + net = Network( + height=height, + width=width, + bgcolor=bgcolor, + font_color=font_color, + directed=False, + notebook=False, + cdn_resources='remote' + ) + + # Configure physics for better layout + net.set_options(""" + { + "nodes": { + "borderWidth": 2, + "borderWidthSelected": 4, + "font": { + "size": 14, + "face": "Arial" + } + }, + "edges": { + "color": { + "inherit": true + }, + "smooth": { + "type": "continuous", + "forceDirection": "none" + } + }, + "physics": { + "enabled": true, + "barnesHut": { + "gravitationalConstant": -30000, + "centralGravity": 0.3, + "springLength": 150, + "springConstant": 0.04, + "damping": 0.09 + }, + "stabilization": { + "enabled": true, + "iterations": 200, + "updateInterval": 25 + } + }, + "interaction": { + "hover": true, + "tooltipDelay": 100, + "zoomView": true, + "dragView": true + } + } + """) + + # Add nodes + for node in subgraph.nodes(): + node_data = subgraph.nodes[node] + is_query_entity = any(qe in node.lower() for qe in query_entities_lower) + + # Determine color + if color_by == "community": + community = node_data.get('community', 0) + color = get_community_color(int(community) if pd.notna(community) else 0) + else: + node_type = node_data.get('type', 'default') + color = get_node_color(node_type, is_query_entity) + + if is_query_entity: + color = "#ff6b6b" # Override for query entities + + # Node size based on degree + degree = subgraph.degree(node) + size = min(10 + degree * 3, 50) + + # Create tooltip + description = node_data.get('description', '') + if len(description) > 200: + description = description[:200] + "..." + + tooltip = f"{node}

{description}" + + net.add_node( + node, + label=node[:30] + "..." if len(node) > 30 else node, + title=tooltip, + color=color, + size=size, + borderWidth=4 if is_query_entity else 2, + font={"size": 12 if is_query_entity else 10} + ) + + # Add edges + for source, target in subgraph.edges(): + edge_data = subgraph.edges[source, target] + weight = edge_data.get('weight', 1) + description = edge_data.get('description', '') + + net.add_edge( + source, + target, + title=description, + width=min(1 + weight * 0.5, 5), + color={"color": "#555555", "opacity": 0.7} + ) + + # Generate HTML + html_content = net.generate_html() + + # Encode the HTML as base64 for data URI embedding in iframe + # This allows Gradio to display full HTML documents + html_bytes = html_content.encode('utf-8') + html_base64 = base64.b64encode(html_bytes).decode('utf-8') + + # Return an iframe with data URI - works in Gradio's HTML component + iframe_html = f''' + +

+ 💡 Tip: Drag nodes to rearrange • Scroll to zoom • Hover for details • Click to select +

+ ''' + + return iframe_html + + +def create_graph_html_for_query( + input_dir: str, + query_entities: List[str] = None, + max_nodes: int = 50, + color_by: str = "community" +) -> str: + """Create a complete HTML visualization for a query.""" + + try: + G = create_full_graph(input_dir) + + if len(G.nodes()) == 0: + return "
No graph data available. Please index your documents first.
" + + subgraph = extract_subgraph_for_query(G, query_entities or [], max_nodes=max_nodes) + html = create_pyvis_graph(subgraph, query_entities, color_by=color_by) + + return html + + except Exception as e: + return f"
Error creating graph: {str(e)}
" + + +def get_graph_stats(input_dir: str) -> Dict[str, Any]: + """Get statistics about the knowledge graph.""" + + try: + entity_df, relationship_df, community_df = load_graph_data(input_dir) + + return { + "total_entities": len(entity_df), + "total_relationships": len(relationship_df), + "total_communities": len(community_df) if len(community_df) > 0 else "N/A", + "entity_types": entity_df['type'].value_counts().to_dict() if 'type' in entity_df.columns else {}, + } + except Exception as e: + return {"error": str(e)} + + +def extract_entities_from_response(response: str, entity_df: pd.DataFrame) -> List[str]: + """Extract entity names mentioned in a response by matching against known entities.""" + + if entity_df is None or len(entity_df) == 0: + return [] + + response_lower = response.lower() + mentioned_entities = [] + + # Get entity names + if 'title' in entity_df.columns: + entity_names = entity_df['title'].tolist() + elif 'name' in entity_df.columns: + entity_names = entity_df['name'].tolist() + else: + return [] + + for entity in entity_names: + if entity and str(entity).lower() in response_lower: + mentioned_entities.append(str(entity)) + + return mentioned_entities[:20] # Limit to top 20 diff --git a/graphrag-ollama-config/ingest.py b/graphrag-ollama-config/ingest.py new file mode 100644 index 0000000..0ea09de --- /dev/null +++ b/graphrag-ollama-config/ingest.py @@ -0,0 +1,898 @@ +""" +Instant Knowledge Ingest Module +Handles YouTube and Web Article URL ingestion for VeritasGraph. +""" + +import os +import re +import subprocess +import hashlib +from datetime import datetime +from urllib.parse import urlparse, parse_qs +from typing import Tuple, Optional +import json + +# YouTube transcript extraction +try: + from youtube_transcript_api import YouTubeTranscriptApi + from youtube_transcript_api._errors import ( + TranscriptsDisabled, + NoTranscriptFound, + VideoUnavailable + ) + YOUTUBE_AVAILABLE = True +except ImportError: + YOUTUBE_AVAILABLE = False + +# Web article extraction +try: + import trafilatura + TRAFILATURA_AVAILABLE = True +except ImportError: + TRAFILATURA_AVAILABLE = False + +# Get the script directory for input folder path +SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) +INPUT_DIR = os.path.join(SCRIPT_DIR, "input") + + +def ensure_input_dir(): + """Ensure input directory exists.""" + os.makedirs(INPUT_DIR, exist_ok=True) + + +def extract_youtube_video_id(url: str) -> Optional[str]: + """ + Extract video ID from various YouTube URL formats. + + Supports: + - https://www.youtube.com/watch?v=VIDEO_ID + - https://youtu.be/VIDEO_ID + - https://www.youtube.com/embed/VIDEO_ID + - https://www.youtube.com/v/VIDEO_ID + """ + if not url: + return None + + # Parse the URL + parsed = urlparse(url) + + # youtu.be format + if parsed.netloc in ['youtu.be', 'www.youtu.be']: + return parsed.path.lstrip('/') + + # youtube.com formats + if parsed.netloc in ['youtube.com', 'www.youtube.com', 'm.youtube.com']: + # Standard watch URL + if parsed.path == '/watch': + query_params = parse_qs(parsed.query) + return query_params.get('v', [None])[0] + + # Embed or v format + if parsed.path.startswith('/embed/') or parsed.path.startswith('/v/'): + return parsed.path.split('/')[2] + + return None + + +def is_youtube_url(url: str) -> bool: + """Check if URL is a YouTube video URL.""" + return extract_youtube_video_id(url) is not None + + +def get_youtube_transcript(video_id: str) -> Tuple[bool, str, Optional[dict]]: + """ + Fetch transcript from a YouTube video. + + Returns: + Tuple of (success, content_or_error, metadata) + """ + if not YOUTUBE_AVAILABLE: + return False, "youtube-transcript-api is not installed. Run: pip install youtube-transcript-api", None + + try: + # New API (v1.x): Instantiate and use fetch method + api = YouTubeTranscriptApi() + transcript_data = None + transcript_type = "auto-detected" + + try: + # Fetch transcript (auto-selects best available) + transcript_data = api.fetch(video_id) + except Exception as e: + error_msg = str(e) + # Check if it's a "no transcripts" situation + if "No transcripts" in error_msg or "disabled" in error_msg.lower(): + return False, f"This video has no transcripts/captions available. Try a video with CC enabled.", None + elif "Video unavailable" in error_msg or "unavailable" in error_msg.lower(): + return False, "Video is unavailable, private, or does not exist.", None + else: + return False, f"Could not fetch transcript: {error_msg}", None + + if transcript_data is None or len(transcript_data) == 0: + return False, "No suitable transcript found for this video. The video may not have captions enabled.", None + + # Combine transcript segments into readable text + # New API returns FetchedTranscriptSnippet objects with .text attribute + full_text = [] + for segment in transcript_data: + # Handle both old dict format and new object format + if hasattr(segment, 'text'): + text = segment.text.strip() + elif isinstance(segment, dict): + text = segment.get('text', '').strip() + else: + text = str(segment).strip() + if text: + full_text.append(text) + + combined_text = ' '.join(full_text) + + # Clean up the text + combined_text = re.sub(r'\s+', ' ', combined_text) # Normalize whitespace + combined_text = combined_text.replace('[Music]', '').replace('[Applause]', '') + combined_text = re.sub(r'\[.*?\]', '', combined_text) # Remove other bracketed content + + if len(combined_text.strip()) < 50: + return False, "Transcript is too short or empty. The video may only have music/non-speech content.", None + + metadata = { + "video_id": video_id, + "transcript_type": transcript_type, + "segment_count": len(transcript_data), + "character_count": len(combined_text), + "word_count": len(combined_text.split()) + } + + return True, combined_text, metadata + + except TranscriptsDisabled: + return False, "Transcripts are disabled for this video by the uploader.", None + except NoTranscriptFound: + return False, "No transcript/captions found for this video.", None + except VideoUnavailable: + return False, "Video is unavailable, private, or does not exist.", None + except Exception as e: + return False, f"Error fetching transcript: {str(e)}", None + + +def get_youtube_metadata(video_id: str) -> dict: + """ + Get basic metadata for a YouTube video using yt-dlp. + + Returns dict with title, description, channel, etc. + """ + try: + import subprocess + result = subprocess.run( + ['yt-dlp', '--dump-json', '--skip-download', f'https://www.youtube.com/watch?v={video_id}'], + capture_output=True, + text=True, + timeout=30 + ) + if result.returncode == 0: + data = json.loads(result.stdout) + return { + "title": data.get("title", "Unknown Title"), + "channel": data.get("channel", data.get("uploader", "Unknown")), + "description": data.get("description", "")[:500], + "duration": data.get("duration", 0), + "upload_date": data.get("upload_date", ""), + "view_count": data.get("view_count", 0) + } + except Exception: + pass + + return {"title": f"YouTube Video {video_id}", "channel": "Unknown"} + + +def extract_web_article(url: str) -> Tuple[bool, str, Optional[dict]]: + """ + Extract main content from a web article URL. + + Returns: + Tuple of (success, content_or_error, metadata) + """ + if not TRAFILATURA_AVAILABLE: + return False, "trafilatura is not installed. Run: pip install trafilatura", None + + try: + # Download the page + downloaded = trafilatura.fetch_url(url) + + if downloaded is None: + return False, "Could not download the webpage. Please check the URL.", None + + # Extract the main content + content = trafilatura.extract( + downloaded, + include_comments=False, + include_tables=True, + include_images=False, + include_links=False, + output_format='txt' + ) + + if not content or len(content.strip()) < 100: + return False, "Could not extract meaningful content from this page. It may be blocked or require JavaScript.", None + + # Try to extract metadata + metadata_result = trafilatura.extract( + downloaded, + output_format='json', + include_comments=False + ) + + metadata = {"url": url} + if metadata_result: + try: + meta_json = json.loads(metadata_result) + metadata.update({ + "title": meta_json.get("title", "Unknown Title"), + "author": meta_json.get("author", "Unknown"), + "date": meta_json.get("date", ""), + "sitename": meta_json.get("sitename", urlparse(url).netloc) + }) + except Exception: + metadata["title"] = urlparse(url).netloc + + metadata["character_count"] = len(content) + metadata["word_count"] = len(content.split()) + + return True, content, metadata + + except Exception as e: + return False, f"Error extracting article: {str(e)}", None + + +def generate_filename(url: str, content_type: str, metadata: Optional[dict] = None) -> str: + """ + Generate a descriptive filename for the ingested content. + + Format: {type}_{safe_title}_{short_hash}.txt + """ + # Create a short hash from the URL for uniqueness + url_hash = hashlib.md5(url.encode()).hexdigest()[:8] + + # Get title from metadata or derive from URL + if metadata and metadata.get("title"): + title = metadata["title"] + elif content_type == "youtube": + video_id = extract_youtube_video_id(url) or "video" + title = f"youtube_{video_id}" + else: + # Use domain + path for articles + parsed = urlparse(url) + title = f"{parsed.netloc}_{parsed.path}" + + # Clean title for filename + safe_title = re.sub(r'[^\w\s-]', '', title) + safe_title = re.sub(r'[-\s]+', '_', safe_title) + safe_title = safe_title[:50] # Limit length + + return f"{content_type}_{safe_title}_{url_hash}.txt" + + +def format_content_for_graphrag(content: str, metadata: dict, source_type: str) -> str: + """ + Format the content with metadata header for better GraphRAG indexing. + """ + header_lines = [ + f"# Source: {source_type.upper()}", + f"# Title: {metadata.get('title', 'Unknown')}", + ] + + if source_type == "youtube": + header_lines.extend([ + f"# Channel: {metadata.get('channel', 'Unknown')}", + f"# Video ID: {metadata.get('video_id', 'Unknown')}", + ]) + if metadata.get('duration'): + duration_min = metadata.get('duration', 0) // 60 + header_lines.append(f"# Duration: {duration_min} minutes") + else: + header_lines.extend([ + f"# URL: {metadata.get('url', 'Unknown')}", + f"# Author: {metadata.get('author', 'Unknown')}", + f"# Site: {metadata.get('sitename', 'Unknown')}", + ]) + + header_lines.extend([ + f"# Ingested: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}", + f"# Word Count: {metadata.get('word_count', 'Unknown')}", + "", + "---", + "" + ]) + + return '\n'.join(header_lines) + content + + +def save_content(filename: str, content: str) -> str: + """ + Save content to the input directory. + + Returns the full path to the saved file. + """ + ensure_input_dir() + filepath = os.path.join(INPUT_DIR, filename) + + with open(filepath, 'w', encoding='utf-8') as f: + f.write(content) + + return filepath + + +def ingest_url(url: str, auto_index: bool = False) -> Tuple[bool, str, Optional[str]]: + """ + Main function to ingest content from a URL. + + Args: + url: YouTube or web article URL + auto_index: If True, automatically trigger GraphRAG indexing + + Returns: + Tuple of (success, message, filepath) + """ + url = url.strip() + + if not url: + return False, "Please provide a URL.", None + + # Validate URL format + if not url.startswith(('http://', 'https://')): + url = 'https://' + url + + # Determine content type and extract + if is_youtube_url(url): + video_id = extract_youtube_video_id(url) + + # Get video metadata first + yt_metadata = get_youtube_metadata(video_id) + + # Get transcript + success, content, transcript_meta = get_youtube_transcript(video_id) + + if not success: + return False, f"❌ YouTube Error: {content}", None + + # Merge metadata + metadata = {**yt_metadata, **(transcript_meta or {})} + content_type = "youtube" + + else: + # Treat as web article + success, content, metadata = extract_web_article(url) + + if not success: + return False, f"❌ Article Error: {content}", None + + content_type = "article" + + # Generate filename and format content + filename = generate_filename(url, content_type, metadata) + formatted_content = format_content_for_graphrag(content, metadata, content_type) + + # Save to input directory + filepath = save_content(filename, formatted_content) + + # Build success message + title = metadata.get('title', 'Unknown') + word_count = metadata.get('word_count', len(content.split())) + + message_parts = [ + f"✅ **Successfully ingested!**", + f"", + f"📄 **Title:** {title}", + f"📊 **Words:** {word_count:,}", + f"💾 **Saved to:** `{filename}`", + f"", + ] + + if content_type == "youtube": + if metadata.get('duration'): + message_parts.insert(3, f"⏱️ **Duration:** {metadata['duration'] // 60} minutes") + message_parts.insert(3, f"📺 **Channel:** {metadata.get('channel', 'Unknown')}") + else: + message_parts.insert(3, f"🌐 **Site:** {metadata.get('sitename', 'Unknown')}") + + message_parts.append("🔄 **Next:** Click 'Index Now' to add this to your knowledge graph!") + + message = '\n'.join(message_parts) + + # Auto-index if requested + if auto_index: + index_success, index_msg = trigger_graphrag_index() + message += f"\n\n{index_msg}" + + return True, message, filepath + + +def ingest_text_content(title: str, content: str, auto_index: bool = False) -> Tuple[bool, str, Optional[str]]: + """ + Ingest raw text content directly (copy-pasted from files, documents, etc.). + + Args: + title: Title for the content (used in filename and metadata) + content: The raw text content to ingest + auto_index: If True, automatically trigger GraphRAG indexing + + Returns: + Tuple of (success, message, filepath) + """ + # Validate inputs + if not title or not title.strip(): + return False, "⚠️ Please provide a title for the content.", None + + if not content or not content.strip(): + return False, "⚠️ Please provide some text content to ingest.", None + + title = title.strip() + content = content.strip() + + # Check minimum content length + if len(content) < 50: + return False, "⚠️ Content is too short. Please provide at least 50 characters of meaningful text.", None + + # Generate filename + content_hash = hashlib.md5(content.encode()).hexdigest()[:8] + safe_title = re.sub(r'[^\w\s-]', '', title) + safe_title = re.sub(r'[-\s]+', '_', safe_title) + safe_title = safe_title[:50] # Limit length + filename = f"text_{safe_title}_{content_hash}.txt" + + # Build metadata + metadata = { + "title": title, + "source": "direct_text_input", + "word_count": len(content.split()), + "character_count": len(content) + } + + # Format content with header + header_lines = [ + f"# Source: DIRECT TEXT INPUT", + f"# Title: {title}", + f"# Ingested: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}", + f"# Word Count: {metadata['word_count']}", + f"# Character Count: {metadata['character_count']}", + "", + "---", + "" + ] + formatted_content = '\n'.join(header_lines) + content + + # Save to input directory + filepath = save_content(filename, formatted_content) + + # Build success message + message_parts = [ + f"✅ **Successfully ingested text content!**", + f"", + f"📄 **Title:** {title}", + f"📊 **Words:** {metadata['word_count']:,}", + f"📝 **Characters:** {metadata['character_count']:,}", + f"💾 **Saved to:** `{filename}`", + f"", + f"🔄 **Next:** Click 'Full Index' or 'Update Index' to add this to your knowledge graph!" + ] + + message = '\n'.join(message_parts) + + # Auto-index if requested + if auto_index: + index_success, index_msg = trigger_graphrag_index() + message += f"\n\n{index_msg}" + + return True, message, filepath + + +def trigger_graphrag_index_async() -> Tuple[bool, str]: + """ + Start GraphRAG indexing process in background. + + Returns: + Tuple of (success, message) + """ + try: + import sys + python_exe = sys.executable + + cmd = [python_exe, "-m", "graphrag.index", "--root", SCRIPT_DIR] + + # Start process in background + process = subprocess.Popen( + cmd, + cwd=SCRIPT_DIR, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True + ) + + # Wait briefly to check for immediate errors + try: + stdout, stderr = process.communicate(timeout=3) + if process.returncode != 0: + return False, f"❌ Indexing failed: {stderr[:500]}" + except subprocess.TimeoutExpired: + pass # Still running, which is expected + + return True, "🔄 **Indexing started in background!**" + + except Exception as e: + return False, f"❌ Error: {str(e)}" + + +def get_indexing_status() -> Tuple[str, bool]: + """ + Check the current indexing status by reading the log file. + + Returns: + Tuple of (status_message, is_complete) + """ + log_path = os.path.join(SCRIPT_DIR, "output", "reports", "indexing-engine.log") + docs_path = os.path.join(SCRIPT_DIR, "output", "artifacts", "create_final_documents.parquet") + entities_path = os.path.join(SCRIPT_DIR, "output", "artifacts", "create_final_entities.parquet") + + if not os.path.exists(log_path): + return "⏳ No indexing log found. Click **Index Now** to start indexing.", False + + try: + with open(log_path, 'r', encoding='utf-8', errors='ignore') as f: + content = f.read() + lines = content.strip().split('\n') + + if not lines: + return "⏳ Indexing starting...", False + + # Check for completion + if "All workflows completed successfully" in content: + # Get stats + doc_count = "unknown" + entity_count = "unknown" + try: + import pandas as pd + if os.path.exists(docs_path): + df = pd.read_parquet(docs_path) + doc_count = len(df) + if os.path.exists(entities_path): + df = pd.read_parquet(entities_path) + entity_count = len(df) + except Exception: + pass + + return f"""✅ **Indexing completed successfully!** + +📊 **Current Index Stats:** +- 📄 Documents: **{doc_count}** +- 🏷️ Entities: **{entity_count}** + +🎉 Ready to query! Switch to the **Chat** tab.""", True + + # Check for errors + if "Error" in lines[-1] or "Exception" in lines[-1]: + return f"❌ Error detected: {lines[-1][:200]}", True + + # Define workflow stages for better progress display + workflow_stages = { + "create_base_text_units": "📝 Chunking text into units", + "create_base_extracted_entities": "🔍 Extracting entities", + "create_summarized_entities": "📋 Summarizing descriptions", + "create_base_entity_graph": "🕸️ Building entity graph", + "create_final_entities": "✨ Finalizing entities", + "create_final_nodes": "📍 Creating graph nodes", + "create_final_communities": "👥 Detecting communities", + "create_final_relationships": "🔗 Finalizing relationships", + "create_final_text_units": "📄 Finalizing text units", + "create_final_community_reports": "📊 Generating community reports", + "create_base_documents": "📚 Processing documents", + "create_final_documents": "✅ Finalizing documents" + } + + # Find latest workflow stage + current_stage = "🔄 Initializing..." + for stage, desc in workflow_stages.items(): + if stage in content: + current_stage = desc + + return f"⏳ **Indexing in progress...**\n\n{current_stage}", False + + except Exception as e: + return f"⚠️ Could not read log: {str(e)}", False + + +def trigger_graphrag_index_with_progress(update_mode: bool = False): + """ + Generator function that triggers GraphRAG indexing and yields progress updates. + + Args: + update_mode: If True, use --update-index flag for incremental indexing + + Yields: + str: Progress messages to display in the UI + """ + import sys + import time + python_exe = sys.executable + + log_path = os.path.join(SCRIPT_DIR, "output", "reports", "indexing-engine.log") + + # Count input files first + input_files = [f for f in os.listdir(INPUT_DIR) if f.endswith('.txt')] + file_count = len(input_files) + + # Build command + cmd = [python_exe, "-m", "graphrag.index", "--root", SCRIPT_DIR, "--reporter", "print"] + + # For update mode, find the last run ID + index_type = "Full" + if update_mode: + last_run_id = get_last_run_id() + if last_run_id: + cmd.extend(["--update-index", last_run_id]) + index_type = "Update" + yield f"🔄 **Starting Update Index**\n\n📁 Found **{file_count}** text files in input folder\n🔗 Using previous run: `{last_run_id}`\n\n⏳ Initializing..." + else: + yield f"🚀 **Starting Full Index** (no previous run found)\n\n📁 Found **{file_count}** text files in input folder\n\n⏳ Initializing..." + else: + # Clear old log for full reindex + if os.path.exists(log_path): + try: + os.remove(log_path) + except Exception: + pass + yield f"🚀 **Starting Full GraphRAG Index**\n\n📁 Found **{file_count}** text files in input folder\n\n⏳ Initializing..." + + try: + # Start process + process = subprocess.Popen( + cmd, + cwd=SCRIPT_DIR, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + bufsize=1 + ) + + max_wait_seconds = 1200 # 20 minutes max + start_time = time.time() + last_log_size = 0 + workflow_stages = { + "create_base_text_units": "📝 Chunking text into units...", + "create_base_extracted_entities": "🔍 Extracting entities (this takes a while)...", + "create_summarized_entities": "📋 Summarizing entity descriptions...", + "create_base_entity_graph": "🕸️ Building entity graph...", + "create_final_entities": "✨ Finalizing entities...", + "create_final_nodes": "📍 Creating graph nodes...", + "create_final_communities": "👥 Detecting communities...", + "create_final_relationships": "🔗 Finalizing relationships...", + "create_final_text_units": "📄 Finalizing text units...", + "create_final_community_reports": "📊 Generating community reports (this takes a while)...", + "create_base_documents": "📚 Processing documents...", + "create_final_documents": "✅ Finalizing documents..." + } + current_stage = "Initializing..." + + while process.poll() is None: + elapsed = time.time() - start_time + if elapsed > max_wait_seconds: + process.terminate() + yield "❌ **Indexing timed out after 20 minutes.** Check logs for issues." + return + + # Read log file for progress + if os.path.exists(log_path): + try: + with open(log_path, 'r', encoding='utf-8', errors='ignore') as f: + content = f.read() + + # Check for workflow stages + for stage, desc in workflow_stages.items(): + if stage in content and desc != current_stage: + current_stage = desc + elapsed_min = elapsed / 60 + yield f"🔄 **{index_type} Indexing in progress** ({elapsed_min:.1f} min)\n\n{current_stage}\n\n📁 Processing {file_count} files..." + + except Exception: + pass + + time.sleep(3) + + # Process completed + stdout_rest, _ = process.communicate() + + if process.returncode == 0: + # Verify success and get stats + docs_path = os.path.join(SCRIPT_DIR, "output", "artifacts", "create_final_documents.parquet") + entities_path = os.path.join(SCRIPT_DIR, "output", "artifacts", "create_final_entities.parquet") + + doc_count = "unknown" + entity_count = "unknown" + + try: + import pandas as pd + if os.path.exists(docs_path): + df = pd.read_parquet(docs_path) + doc_count = len(df) + if os.path.exists(entities_path): + df = pd.read_parquet(entities_path) + entity_count = len(df) + except Exception: + pass + + elapsed_total = (time.time() - start_time) / 60 + index_emoji = "🔄" if update_mode else "📊" + yield f"""✅ **{index_emoji} {index_type} Indexing completed successfully!** + +📊 **Results:** +- 📄 Documents indexed: **{doc_count}** +- 🏷️ Entities extracted: **{entity_count}** +- ⏱️ Time taken: **{elapsed_total:.1f} minutes** + +🎉 Your knowledge graph is now ready! Switch to the **Chat** tab to query your data.""" + else: + yield f"❌ **Indexing failed** (exit code: {process.returncode})\n\nCheck the logs at:\n`output/reports/indexing-engine.log`" + + except FileNotFoundError: + yield "❌ **GraphRAG not found.** Make sure graphrag is installed." + except Exception as e: + yield f"❌ **Error:** {str(e)}" + + +def get_last_run_id() -> Optional[str]: + """ + Extract the last run ID from the indexing log file. + + Returns: + The run ID (e.g., '20260113-090657') or None if not found + """ + log_path = os.path.join(SCRIPT_DIR, "output", "reports", "indexing-engine.log") + + if not os.path.exists(log_path): + return None + + try: + with open(log_path, 'r', encoding='utf-8', errors='ignore') as f: + for line in f: + # Look for: "Starting pipeline run for: 20260113-090657" + if "Starting pipeline run for:" in line: + match = re.search(r'run for:\s*(\d{8}-\d{6})', line) + if match: + return match.group(1) + except Exception: + pass + + return None + + +def trigger_graphrag_index(update_mode: bool = False) -> Tuple[bool, str]: + """ + Trigger GraphRAG indexing process and wait for completion. + + Args: + update_mode: If True, use --update-index flag for incremental indexing + + Returns: + Tuple of (success, message) + """ + try: + import sys + import time + python_exe = sys.executable + + # Build command + cmd = [python_exe, "-m", "graphrag.index", "--root", SCRIPT_DIR, "--reporter", "print"] + + # For update mode, find the last run ID and use --update-index + if update_mode: + last_run_id = get_last_run_id() + if last_run_id: + cmd.extend(["--update-index", last_run_id]) + else: + # No previous run found, fall back to full index + update_mode = False + + # Clear old log for fresh progress tracking + log_path = os.path.join(SCRIPT_DIR, "output", "reports", "indexing-engine.log") + + # Start process + process = subprocess.Popen( + cmd, + cwd=SCRIPT_DIR, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True + ) + + # Wait for process to complete (with timeout) + max_wait_seconds = 1200 # 20 minutes max + start_time = time.time() + + while process.poll() is None: + elapsed = time.time() - start_time + if elapsed > max_wait_seconds: + process.terminate() + return False, "❌ Indexing timed out after 20 minutes. Check logs for issues." + time.sleep(2) + + # Process completed + stdout, stderr = process.communicate() + + if process.returncode == 0: + # Verify by checking log + if os.path.exists(log_path): + with open(log_path, 'r', encoding='utf-8', errors='ignore') as f: + log_content = f.read() + if "All workflows completed successfully" in log_content: + # Count documents + docs_path = os.path.join(SCRIPT_DIR, "output", "artifacts", "create_final_documents.parquet") + entities_path = os.path.join(SCRIPT_DIR, "output", "artifacts", "create_final_entities.parquet") + doc_count = "unknown" + entity_count = "unknown" + try: + import pandas as pd + df = pd.read_parquet(docs_path) + doc_count = len(df) + df = pd.read_parquet(entities_path) + entity_count = len(df) + except Exception: + pass + + index_type = "🔄 Update" if update_mode else "📊 Full" + return True, f"✅ **{index_type} Indexing completed successfully!**\n\n📄 **Documents indexed:** {doc_count}\n🏷️ **Entities:** {entity_count}\n\n🎉 You can now chat with your knowledge graph!" + + return True, "✅ **Indexing completed!** Refresh the page to query your new data." + else: + error_msg = stderr[:500] if stderr else "Unknown error" + return False, f"❌ Indexing failed:\n\n```\n{error_msg}\n```" + + except FileNotFoundError: + return False, "❌ GraphRAG not found. Make sure graphrag is installed." + except Exception as e: + return False, f"❌ Error: {str(e)}" + + +def list_input_files() -> list: + """List all files in the input directory.""" + ensure_input_dir() + files = [] + for f in os.listdir(INPUT_DIR): + if f.endswith('.txt'): + filepath = os.path.join(INPUT_DIR, f) + stat = os.stat(filepath) + files.append({ + "name": f, + "size": stat.st_size, + "modified": datetime.fromtimestamp(stat.st_mtime).strftime('%Y-%m-%d %H:%M') + }) + return sorted(files, key=lambda x: x['modified'], reverse=True) + + +def delete_input_file(filename: str) -> Tuple[bool, str]: + """Delete a file from the input directory.""" + filepath = os.path.join(INPUT_DIR, filename) + if os.path.exists(filepath): + os.remove(filepath) + return True, f"✅ Deleted: {filename}" + return False, f"❌ File not found: {filename}" + + +def get_file_preview(filename: str, max_chars: int = 1000) -> str: + """Get a preview of file contents.""" + filepath = os.path.join(INPUT_DIR, filename) + if os.path.exists(filepath): + with open(filepath, 'r', encoding='utf-8') as f: + content = f.read(max_chars) + if len(content) == max_chars: + content += "\n\n... (truncated)" + return content + return "File not found" + + +# Check dependencies on module load +def check_dependencies() -> dict: + """Check which optional dependencies are available.""" + return { + "youtube": YOUTUBE_AVAILABLE, + "web": TRAFILATURA_AVAILABLE + } diff --git a/graphrag-ollama-config/input/article_Unknown_Title_d8de84d9.txt b/graphrag-ollama-config/input/article_Unknown_Title_d8de84d9.txt new file mode 100644 index 0000000..201d0cb --- /dev/null +++ b/graphrag-ollama-config/input/article_Unknown_Title_d8de84d9.txt @@ -0,0 +1,30 @@ +# Source: ARTICLE +# Title: Unknown Title +# URL: https://ocean.edu.vn/vi-VN/gioi-thieu/gioi-thieu-ve-ocean-edu-4 +# Author: Unknown +# Site: ocean.edu.vn +# Ingested: 2026-01-31 11:25:26 +# Word Count: 973 + +--- +1. Sự hình thành và phát triển của Ocean Edu +Hệ thống Anh ngữ Quốc tế Ocean Edu là một trong những tổ chức giáo dục Anh ngữ uy tín tại Việt Nam, cung cấp các chương trình đào tạo tiếng Anh theo chuẩn quốc tế, đáp ứng nhu cầu học tập đa dạng cho nhiều độ tuổi. Sau 19 năm hình thành và phát triển, Ocean Edu đã xây dựng mạng lưới gần 200 chi nhánh đào tạo phủ khắp các tỉnh, thành trên toàn quốc. +Ocean Edu sở hữu đội ngũ 100% giáo viên nước ngoài có trình độ chuyên môn sư phạm cao, giàu kinh nghiệm và tận tâm với nghề, đồng hành cùng hơn 2.000 cán bộ, chuyên viên chuyên nghiệp và hệ thống cơ sở vật chất hiện đại, đồng bộ. Mỗi năm, Ocean Edu chào đón hơn 150.000 lượt học viên và đã đào tạo thành công trên 1.000.000 học viên trên toàn quốc. Đặc biệt, hơn 90% học viên tiếp tục quay lại học tập, khẳng định chất lượng đào tạo hiệu quả và giá trị bền vững mà Ocean Edu không ngừng theo đuổi. +Với tầm nhìn dài hạn và khát vọng phát triển, Ocean Edu luôn nỗ lực trở thành tập đoàn giáo dục được hàng triệu người tin yêu và lựa chọn tại Việt Nam, đóng góp tích cực vào việc nâng cao chất lượng giảng dạy tiếng Anh, góp phần thay đổi diện mạo giáo dục ngoại ngữ và nâng tầm vị thế của người Việt trên trường quốc tế. +Lộ trình và mục tiêu phát triển Ocean Edu đến năm 2025 +Trên hành trình phát triển đó, Ocean Edu đã vinh dự nhận được nhiều danh hiệu và giải thưởng uy tín, ghi nhận những đóng góp nổi bật và vị thế của một thương hiệu trung tâm Anh ngữ mang tầm vóc quốc tế tại Việt Nam. (Xem giải thưởng và danh hiệu tại đây) +19 năm – một hành trình bền bỉ và tiên phong, Ocean Edu kiên định với sứ mệnh “Giúp hàng triệu người Việt Nam giỏi tiếng Anh”, tự hào là nơi khơi mở tiềm năng, đồng hành cùng hàng triệu học viên trên con đường chinh phục giấc mơ trở thành công dân toàn cầu. +Ocean Edu 18 năm một chặng đường +2. Mục tiêu và triết lý giáo dục +3. Logo và Slogan của OCEAN EDU +Logo: Sự kết tinh giữa hình ảnh ngọn hải đăng và vòng nguyệt quế cùng với gam màu xanh của trời và biển đã tạo nên hình ảnh logo vô cùng hài hòa mà ấn tượng, tạo cảm giác bình yên nhưng vững chắc như khẳng định những con tàu tri thức vượt muôn trùng đại dương sẽ cập bến thành công trong niềm vui chiến thắng và vinh quang. +Slogan: “Turn on your potential - thắp sáng tiềm năng của bạn" +Đây là thông điệp mà Ocean Edu muốn truyền tải về những giá trị thiết thực mỗi học viên đạt được trong thế giới tri thức nhân loại. Ocean Edu sẽ luôn tận tâm thực hiện sứ mệnh của mình trên hành trình đào tạo những nhân tài tiếng Anh cho đất Việt để sẵn sàng vươn mình ra thế giới. Bằng những đóng góp tích cực và hữu ích cho cộng đồng, Ocean Edu tự hào là thương hiệu giáo dục uy tín, được nhiều người tin yêu và lựa chọn. +4. Năm tiêu chí phát triển giáo dục tại Ocean Edu +5. Chương trình học (Khóa học) +Các chương trình học tại Ocean Edu đều được biên soạn và liên tục cập nhật theo tiêu chuẩn quốc tế, mang tính thực tiễn cao, phù hợp với từng lứa tuổi, giúp học viên có thể đạt được hiệu quả cao nhất, áp dụng thành thạo trong công việc và cuộc sống. +Điểm ưu việt trong chất lượng chương trình học tại Ocean Edu đó là việc áp dụng công nghệ và hệ thống quản lý học tập trực tuyến hiện đại, không chỉ giúp quản lý học viên, giáo trình hiệu quả mà còn đẩy mạnh kết nối và tăng cường tương tác giữa phụ huynh, giáo viên mọi lúc, mọi nơi. +6. Học viên +Ocean Edu xây dựng lộ trình học rõ ràng và lâu dài, cam kết đạt hiệu quả, định hướng một tương lai rộng mở cho mỗi học viên. Mỗi em học sinh từ độ tuổi nhỏ cho đến khi bước vào trung học, đại học, và các cấp độ học cao hơn nữa, đều có được sự trải nghiệm và định hướng phát triển cùng môi trường giáo dục quốc tế, giúp học viên hội nhập trở thành công dân toàn cầu. +Ocean Edu sử dụng quy trình học tập chặt chẽ và bài bản: Trước khi đăng ký học, học viên sẽ được làm bài kiểm tra đầu vào để đánh giá trình độ và tư vấn lộ trình học tập phù hợp, trong suốt quá trình học sẽ có các bài kiểm tra giữa kì và cuối kì để theo dõi kết quả học tập của học viên. Bên cạnh đó, học viên còn được tham gia nhiều hoạt động ngoại khóa hấp dẫn. Phụ huynh sẽ liên tục được cập nhật tình hình học tập cũng như định hướng học tập của con trên lớp và các nhận xét của giáo viên. +Với sự chú trọng và quan tâm về chất lượng đào tạo, quy chuẩn quy trình từ con người đến cơ sở vật chất cùng với sự hợp tác của mỗi học viên và phụ huynh, Ocean Edu cam kết sẽ đem đến những sản phẩm, dịch vụ giáo dục chất lượng cao, không ngừng làm hài lòng khách hàng và phụng sự cộng đồng. \ No newline at end of file diff --git a/graphrag-ollama-config/input/dulce.txt b/graphrag-ollama-config/input/dulce.txt new file mode 100644 index 0000000..86975d3 --- /dev/null +++ b/graphrag-ollama-config/input/dulce.txt @@ -0,0 +1,185 @@ +# Operation: Dulce + +## Chapter 1 + +The thrumming of monitors cast a stark contrast to the rigid silence enveloping the group. Agent Alex Mercer, unfailingly determined on paper, seemed dwarfed by the enormity of the sterile briefing room where Paranormal Military Squad's elite convened. With dulled eyes, he scanned the projectors outlining their impending odyssey into Operation: Dulce. + +“I assume, Agent Mercer, you’re not having second thoughts?” It was Taylor Cruz’s voice, laced with an edge that demanded attention. + +Alex flickered a strained smile, still thumbing his folder's corner. "Of course not, Agent Cruz. Just trying to soak in all the details." The compliance in his tone was unsettling, even to himself. + +Jordan Hayes, perched on the opposite side of the table, narrowed their eyes but offered a supportive nod. "Details are imperative. We’ll need your clear-headedness down there, Mercer." + +A comfortable silence, the kind that threaded between veterans of shared secrets, lingered briefly before Sam Rivera, never one to submit to quiet, added, "I’ve combed through the last transmission logs. If anyone can make sense of the anomalies, it’s going to be the two of you." + +Taylor snorted dismissively. “Focus, people. We have protocols for a reason. Speculation is counter-productive.” The words 'counter-productive' seemed to hang in the air, a tacit reprimand directed at Alex. + +Feeling the weight of his compliance conflicting with his natural inclination to leave no stone unturned, Alex straightened in his seat. "I agree, Agent Cruz. Protocol is paramount," he said, meeting Taylor's steely gaze. It was an affirmation, but beneath it lay layers of unspoken complexities that would undoubtedly unwind with time. + +Alex's submission, though seemingly complete, didn't escape Jordan, who tilted their head ever so slightly, their eyes revealing a spark of understanding. They knew well enough the struggle of aligning personal convictions with overarching missions. As everyone began to collect their binders and prepare for departure, a quiet resolve took form within Alex, galvanized by the groundwork laid by their interactions. He may have spoken in compliance, but his determination had merely taken a subtler form — one that wouldn't surrender so easily to the forthcoming shadows. + +\* + +Dr. Jordan Hayes shuffled a stack of papers, their eyes revealing a tinge of skepticism at Taylor Cruz's authoritarian performance. _Protocols_, Jordan thought, _are just the framework, the true challenges we're about to face lie well beyond the boundaries of any protocol._ They cleared their throat before speaking, tone cautious yet firm, "Let's remember, the unknown variables exceed the known. We should remain adaptive." + +A murmur of agreement echoed from Sam Rivera, who leaned forward, lacing their fingers together as if weaving a digital framework in the air before them, "Exactly, adaptability could be the key to interpreting the signal distortions and system malfunctions. We shouldn't discount the… erratic." + +Their words hung like an electric charge in the room, challenging Taylor's position with an inherent truth. Cruz’s jaw tightened almost imperceptibly, but the agent masked it with a small nod, conceding to the omnipresent threat of the unpredictable. + +Alex glanced at Jordan, who never looked back, their gaze fixed instead on a distant point, as if envisioning the immense dark corridors they were soon to navigate in Dulce. Jordan was not one to embrace fantastical theories, but the air of cautious calculation betrayed a mind bracing for confrontation with the inexplicable, an internal battle between the evidence of their research and the calculating skepticism that kept them alive in their field. + +The meeting adjourned with no further comments, the team members quietly retreading the paths to their personal preparations. Alex, trailing slightly behind, observed the others. _The cautious reserve Jordan wears like armor doesn't fool me_, he thought, _their analytical mind sees the patterns I do. And that's worth more than protocol. That's the connection we need to survive this._ + +As the agents dispersed into the labyrinth of the facility, lost in their thoughts and preparations, the base's halogen lights flickered, a brief and unnoticed harbingers of the darkness to come. + +\* + +A deserted corridor inside the facility stretched before Taylor Cruz, each footstep rhythmic and precise. Cruz, ambitious and meticulous, eyed the troops passing by with a sardonic tilt of the lips. Obedience—it was as much a tool as any weapon in the arsenal, and Cruz wielded it masterfully. To them, it was another step toward unfettered power within the dark bowels of the military complex. + +Inside a secluded equipment bay, Cruz began checking over gear with mechanical efficiency. They traced fingers over the sleek surface of an encrypted radio transmitter. "If protocols are maintained," said Cruz aloud, rehearsing the speech for their subordinates, "not only will we re-establish a line of communication with Dulce, but we shall also illuminate the darkest secrets it conceals." + +Agent Hayes appeared in the doorway, arms crossed and a knowing glint in their eyes. "You do understand," Jordan began, the words measured and probing, "that once we're in the depths, rank gives way to survival instincts. It's not about commands—it's empowerment through trust." + +The sentiment snagged on Cruz's armor of confidence, probing at the insecurities festering beneath. Taylor offered a brief nod, perhaps too curt, but enough to acknowledge Jordan's point without yielding ground. "Trust," Cruz mused, "or the illusion thereof, is just as potent." + +Silence claimed the space between them, steeped in the reality of the unknown dangers lurking in the shadows of the mission. Cruz diligently returned to the equipment, the act a clear dismissal. + +Not much later, Cruz stood alone, the hollow echo of the bay a stark reminder of the isolation that power often wrought. With each checked box, their resolve steeled further, a silent vow to usher their team through the abyss—whatever it might hold—and emerge enshrined in the respect they so deeply craved. + +## Chapter 2 + +Sam Rivera sat alone in a cramped office, the hum of a dozen servers murmuring a digital lullaby in the background. Surrounded by the glow of multiple screens, their eyes danced across lines of code and intercepted comm signals from Dulce — a kaleidoscope of data that their curious and isolated mind hungered to decrypt. + +To an outsider, it might have looked like obsession, this fervent quest for answers. But to Sam, it was a dance — a give and take with the mysteries of the universe. Their fingers paused over the keyboard as they leaned back in the chair, whispering to thin air, "What secrets are you hiding from us?" + +The stillness of the room broke with the unexpected arrival of Alex Mercer, whose encroaching shadow loomed over Sam's workspace. The cybersecurity expert craned their neck upwards, met by the ever-so-slight furrow in Alex's brow. "Got a minute, Rivera?" + +"Always," Sam said, a smile surfacing as they swiveled to face their mentor more directly. _He has that look — like something's not sitting right with him,_ they noted inwardly. + +Alex hesitated, weighing his words carefully. "Our tech is top-tier, but the silence from Dulce... It's not just technology that will see us through, it's intuition and... trust." His gaze pierced through the digital haze, trying to instill something more profound than advice. + +Sam regarded Alex for a moment, the sincerity in his voice resonating with their own unspoken desire to prove their worth. "Intuition," they mirrored thoughtfully. "I guess sometimes the numbers don't have all the answers." + +Their shared silence held a newfound understanding, a recognition that between the ones and zeros, it was their combined human insights that might prevail against the impossible. As Alex turned to leave, Sam's eyes drifted back to the screens, now seeing them not as barriers to isolate behind, but as windows into the vast and enigmatic challenge that awaited their team. + +Outside the office, the persistent buzz of activity in the facility belied the unease that gripped its inhabitants. A restlessness that nibbled on the edges of reality, as though forewarning of the threshold they were soon to cross — from the known into the realm of cosmic secrets and silent threats. + +\* + +Shadows played against the walls of the cramped underground meeting room, where Alex Mercer stood gazing at the concealed elevator that would deliver them into the bowels of Dulce base. The air was thick, every breath laced with the weight of impending confrontation, the kind one feels when stepping into a legend. Though armed with an array of advanced weaponry and gear, there was an unshakeable sense that they were delving into a conflict where the physical might be of little consequence. + +"I know what you're thinking," Jordan Hayes remarked, approaching Mercer. Their voice was low, a blend of confidence and hidden apprehension. "This feels like more than a rescue or reconnaissance mission, doesn't it?" + +Alex turned, his features a mask of uneasy resolve. "It's like we're being pulled into someone else’s game. Not just observers or participants, but... pawns." + +Jordan gave a short nod, their analytical mind colliding with the uncertain dynamics of this operation. "I've felt that way since the briefing. Like there's a layer we’re not seeing. And yet, we have no choice but to play along." Their eyes locked with Alex's, silently exchanging a vow to remain vigilant. + +"You two need to cut the philosophical chatter. We have positions to secure," Taylor Cruz interjected sharply, stepping into their exchange. The authority in Taylor's voice brooked no argument; it was their way of pulling everyone back to the now. + +Alex's response was measured, more assertive than moments ago. "Acknowledged, Agent Cruz," he replied, his voice steadier, mirroring the transformation brewing within. He gripped his rifle with a newfound firmness. "Let's proceed." + +As they congregated at the elevator, a tension palpable, Sam Rivera piped in with a tone of balanced levity, "Hope everyone’s brought their good luck charms. Something tells me we’re going to need all the help we can get." + +Their laughter served as a brief respite from the gravity of their mission, a shared moment that reinforced their common humanity amidst the unknowable. Then, as one, they stepped into the elevator. The doors closed with a silent hiss, and they descended into the darkness together, aware that when they returned, if they returned, none of them would be the same. + +\* + +The sense of foreboding hung heavier than the darkness that the artificial lights of the elevator shaft failed to fully penetrate. The team was descending into the earth, carrying with them not only the weight of their equipment but also the silent pressure of the invisible war they were about to fight—a war that seemed to edge away from physicality and into the unnervingly psychological. + +As they descended, Dr. Jordan Hayes couldn't help but muse over the layers of data that could wait below, now almost longing for the comfort of empirical evidence. _To think that this reluctance to accept other possibilities may have been my biggest blind spot,_ Jordan contemplated, feeling the hard shell of skepticism begin to crack. + +Alex caught Jordan's reflective gaze and leaned in, his voice barely a murmur over the hum of the elevator. "Once we're down there, keep that analytical edge sharp. You see through the mazes of the unexplained better than anyone." + +The compliment was unexpected and weighed differently than praise from others. This was an acknowledgment from someone who stood on the front lines of the unknown with eyes wide open. "Thank you, Alex," Jordan said, the words carrying a trace of newfound assertiveness. "You can count on me." + +The exchange was cut short by a shudder that ran through the elevator, subtle, but enough to make them instinctively hold their breaths. It wasn't the mechanical stutter of old gears but a vibration that seemed to emanate from the very walls of the shaft—a whisper of something that defied natural explanation. + +Cruz was the first to react, all business despite the shadow that crossed their expression. "Systems check. Now," they barked out, masking the moment of disquiet with swift command. + +Every agent checked their gear, sending confirmation signals through their comms, creating a chorus of electronic beeps that promised readiness. But there was an unspoken question among them: was their technology, their weaponry, their protocols sufficient for what awaited them or merely a fragile comfort? + +Against the gravity of the silence that was once again closing in, Sam's voice crackled through, only half-jest. "I'd laugh if we run into Martians playing poker down there—just to lighten the mood, you know?" + +Despite—or perhaps because of—the oddity of the moment, this elicited a round of chuckles, an audible release of tension that ran counterpoint to the undercurrent of anxiety coursing through the team. + +As the elevator came to a halting, eerie calm at the sub-level, the group stepped off, finding themselves at the threshold of Dulce's mysterious halls. They stood in a tight pack, sharing a cautious glance before fanning out into the unknown, each one acutely aware that the truth was inevitably intertwined with danger. + +Into the depths of Dulce, the team advanced, their silence now a shared testament to the camaraderie born of facing the abyss together—and the steel resolve to uncover whatever horrors lay hidden in its shadows. + +\* + +The weight of the thick metal door closing behind them reverberated through the concrete hallway, marking the final threshold between the familiar world above and the strangeness that lay beneath. Dulce base, a name that had been whispered in the wind-blown deserts above and in the shadowed corners of conspiracy forums, now a tangible cold reality that they could touch — and that touched them back with a chill. + +Like lambs led to an altar of alien deities, so did Agents Alex Mercer, Jordan Hayes, Taylor Cruz, and Sam Rivera proceed, their movements measured, their senses heightened. The air was still, almost respectful of the gravity of their presence. Their torch beams sliced through the darkness, uncovering steel doors with warnings that spoke of top secrets and mortal dangers. + +Taylor Cruz, stepping firmly into the role of de facto leader, set a brisk pace. "Eyes sharp, people. Comms check, every thirty seconds," Taylor ordered, their voice echoing slightly before being swallowed by the surrounding silence. + +Sam, fiddling with a handheld device aimed at detecting electronic anomalies, offered a murmured "Copy that," their usual buoyancy dimmed by the oppressive atmosphere. + +It was Jordan Hayes who paused at an innocuous looking panel, nondescript amongst the gauntlet of secured doorways. "Mercer, Rivera, come see this," Jordan’s voice was marked with a rare hint of urgency. + +Alex joined Jordan's side, examining the panel which, at a mere glance, seemed just another part of the base's infrastructure. Yet, to the trained eye, it appeared out of place—a facade. + +Jordan explained their reasoning as Sam approached, instinctively understanding the significance of what lay beneath, "This panel is a recent addition — covering something they didn't want found." + +Before Alex could respond, the soft whir of an approaching drone cut through their muffled exchange. Taylor had looped back upon hearing the commotion. "Explanations later. We can't afford to attract..." Cruz’s voice trailed off as the small airborne device came into view, its sensors locked onto the group. + +Sam was the first to react, their tech-savvy mind already steps ahead. "I've got this," they declared, fingers flying over the controls of their own gadgetry to ward off the impending threat. + +The drone lingered, its scan seeming more curious than hostile. But within moments, courtesy of Sam's interference, the little sentinel drifted away, retreating into the shadows as if accepting a silent truce. The crew exhaled, a moment of collective relief palpable in the air. + +Cruz squared their shoulders, clearly ruffled but not conceding any ground. "Move out," they directed, a hint more forceful than before. "And Rivera, keep that trick handy." + +The team pressed onward, the quiet now filled with the soft beeps of regular comms checks, their pace undeterred by the confrontation. Yet, every agent held a renewed sense of wariness, their trust in one another deepening with the knowledge that the base—its technology, its secrets—was alive in a way they hadn't fully anticipated. + +As they converged upon a central hub, the imposing doors to the mainframe room stood ajar — an invitation or a trap, neither option comforting. Without a word, they fortified their resolve and stepped through the threshold, where the dim glow of operational LED lights and the distant hum of machinery hinted at Dulce’s still-beating heart. + +Solemnly, yet unmistakably together, they moved deeper into the heart of the enigma, ready to unmask the lifeforce of Dulce base or confront whatever existential threat lay in wait. It was in that unwavering march towards the unknown that their destinies were forever cemented to the legacy of Operation: Dulce. + +## Chapter 3 + +The thrumming of monitors cast a stark contrast to the rigid silence enveloping the group. Agent Alex Mercer, unfailingly determined on paper, seemed dwarfed by the enormity of the sterile briefing room where Paranormal Military Squad's elite convened. With dulled eyes, he scanned the projectors outlining their impending odyssey into Operation: Dulce. + +\* + +The cooling vents hummed in a monotonous drone, but it was the crackle of the comms system coming to life that cut through the lab’s tension. Dr. Jordan Hayes hovered over a table arrayed with alien technology, their fingers delicately probing the enigmatic circuitry retrieved from the crash site. Agent Alex Mercer watched, admiration blooming in silent solidarity for Jordan's deft touch and unspoken drive. + +Jordan, always composed, only allowed the faintest furrow of concentration to mar their brow. "What we understand about physics..." they muttered, trailing off as they realigned a translucent component. The device emitted a low pulse, causing Jordan to still. "Could be fundamentally changed by this." + +A calculated risk—that's what this was. And for a person of science, a gamble was worth the potential paradigm shift. + +"I’ve been thinking," Alex started, his eyes still fixed on the immediately tangible mystery before them. "About what’s at stake here. Not the mission parameters, but what this means for us—humanity." + +Jordan glanced up, meeting his eyes just long enough to convey the shared enormity of their situation; the career-defining glory and existential dread entwined. "The quest for understanding always comes at a price. We're standing on the precipice of knowledge that could either elevate us or condemn us." + +The charged air between them spiked as Taylor Cruz’s brusque tones sliced through their reverie. "Hayes, Mercer, this isn't philosophy hour. Focus on the task. We need actionable intel, not daydreams." + +With a sound of restrained acknowledgment, Jordan returned their gaze to the device, while Alex clenched his jaw, the buzz of frustration dull against the backdrop of Taylor's authoritarian certainty. It was this competitive undercurrent that kept him alert, the sense that his and Jordan's shared commitment to discovery was an unspoken rebellion against Cruz's narrowing vision of control and order. + +Then Taylor did something unexpected. They paused beside Jordan and, for a moment, observed the device with something akin to reverence. “If this tech can be understood..." Taylor said, their voice quieter, "It could change the game for us. For all of us.” + +The underlying dismissal earlier seemed to falter, replaced by a glimpse of reluctant respect for the gravity of what lay in their hands. Jordan looked up, and for a fleeting heartbeat, their eyes locked with Taylor's, a wordless clash of wills softening into an uneasy truce. + +It was a small transformation, barely perceptible, but one that Alex noted with an inward nod. They had all been brought here by different paths and for different reasons. Yet, beneath the veneer of duty, the enticement of the vast unknown pulled them inexorably together, coalescing their distinct desires into a shared pulse of anticipation. + +Marshaled back to the moment by the blink of lights and whir of machinery, they refocused their efforts, each movement sharpened by the knowledge that beyond understanding the unearthly artifacts, they might be piecing together the future of their species. + +\* + +Amidst the sterility of the briefing room, the liminal space between the facts laid out and the hidden truths, sat Sam Rivera, his demeanor an artful balance of focus and a casual disguise of his razor-sharp talent with technology. Across from him, Alex Mercer lingered in thought, the mental cogs turning as each file on Dulce stirred more than curiosity—it beckoned to a past both honored and burdensome. + +"You've been quiet, Sam," Alex noted, catching the younger man's contemplative gaze. "Your take on these signal inconsistencies?" + +There was a respect in Alex's tone, though a respectful distance remained—a gulf of experience and a hint of protective mentorship that stood between them. Sam nodded, recognizing the space afforded to him, and he couldn't help but feel the weight of expectation pressing upon his shoulders. It wasn't just the mission that was immense, it was the trust being placed in him. + +"The patterns are... off," Sam admitted, hesitant but driven. "If I'm right, what we're looking at isn't random—it's a structured anomaly. We need to be ready for anything." + +Alex's eyes brightened with a subtle approval that crossed the distance like a silent nod. "Good. Keen eyes will keep us ahead—or at least not blindsided," he said, affirming the belief that inscribed Sam's role as more than the tech personnel—he was to be a guiding intellect in the heart of uncertainty. + +Their exchange was cut short by Taylor Cruz's abrupt arrival, his gait brimming with a robust confidence that veiled the sharp undercurrents of his striving nature. "Time to gear up. Dulce waits for no one," Taylor announced, his voice carrying an iron resolve that knew the costs of hesitation—though whether the cost was calculated in human or career terms was an ambiguity he wore like a badge of honor. + +As Sam and Alex nodded in unison, the icy chasm of hierarchy and cryptic protocols seemed momentarily to bridge over with an understanding—this mission was convergence, a nexus point that would challenge each of their motives and strength. + +They filed out of the briefing room, their footsteps synchronized, a rhythm that spoke volumes of the unknown cadence they would soon march to within the base's veins. For Alex Mercer, the link with Sam Rivera, though distant, was now poised with a mutuality ready to be tested; for Taylor Cruz, the initiative pulsed like a heartbeat, anticipation thinly veiled behind a mask of duty. + +In the midst of the descent, they were each alone yet irrevocably joined, stepping closer towards the volatile embrace of Operation: Dulce. \ No newline at end of file diff --git a/graphrag-ollama-config/input/elite_athlete_recruitment.txt b/graphrag-ollama-config/input/elite_athlete_recruitment.txt new file mode 100644 index 0000000..d060eb1 --- /dev/null +++ b/graphrag-ollama-config/input/elite_athlete_recruitment.txt @@ -0,0 +1,470 @@ +# Elite Athlete Recruitment: FIFA, UEFA, and UK GBE Compliance Guide + +## Introduction + +This comprehensive guide covers the regulatory frameworks governing elite athlete recruitment across international football (soccer) and UK Governing Body Endorsement requirements for sports professionals. + +## Part 1: FIFA Player Transfer Regulations + +### 1.1 Overview of FIFA Transfer System + +FIFA (Fédération Internationale de Football Association) regulates international player transfers through: + +- FIFA Regulations on the Status and Transfer of Players (RSTP) +- Transfer Matching System (TMS) +- International clearance procedures +- Protection of minors provisions + +### 1.2 Transfer Windows + +**Summer Transfer Window:** +- Primary transfer period +- Duration: Typically 12-16 weeks +- Start: End of domestic season +- End: Before league season start +- Most high-value transfers occur in this period + +**Winter Transfer Window:** +- Secondary transfer period +- Duration: Maximum 4 weeks +- Usually January in most leagues +- Limited activity compared to summer + +### 1.3 International Transfer Certificate (ITC) + +For international transfers, clubs must obtain: + +**ITC Requirements:** +- Request through FIFA TMS +- Previous club approval +- National association clearance +- No outstanding contractual disputes +- Player registration completed + +**Processing Time:** +- Standard processing: 7-30 days +- Urgent cases: 48-72 hours +- Disputes may cause delays + +### 1.4 Training Compensation + +FIFA mandates training compensation for: + +**Eligibility:** +- Players ages 12-23 transferring internationally +- Professional contracts signed before age 23 +- Training clubs entitled to compensation + +**Calculation:** +- Based on training costs by federation category +- Category 1 clubs (highest): €90,000 per year +- Category 2 clubs: €60,000 per year +- Category 3 clubs: €30,000 per year +- Category 4 clubs: €10,000 per year + +### 1.5 Solidarity Mechanism + +FIFA solidarity contribution: +- 5% of transfer fee distributed to training clubs +- Distributed based on years of training +- Ages 12-23 eligible for payments +- 0.25% per year ages 12-15 +- 0.5% per year ages 16-23 + +### 1.6 Protection of Minors + +FIFA strictly regulates minor transfers: + +**General Prohibition:** +- International transfers of players under 18 prohibited +- Except for specific exceptions + +**Exceptions for Minor Transfers:** +1. Parents move to destination country for non-football reasons +2. Transfer within EU/EEA for players aged 16-18 +3. Player lives within 50km of national border +4. First registration after arriving as unaccompanied minor + +**Documentation Required:** +- Proof of family relocation +- Parental consent +- Welfare arrangements +- Educational plans +- Living arrangements + +## Part 2: UEFA Financial Fair Play (FFP) + +### 2.1 Purpose of Financial Fair Play + +UEFA FFP aims to: +- Ensure financial stability of clubs +- Encourage living within means +- Reduce wage and transfer inflation +- Protect long-term viability of football + +### 2.2 Break-Even Requirement + +Clubs must balance football-related: +- Revenue: Broadcasting, matchday, commercial, player sales +- Expenses: Wages, transfers, operating costs + +**Acceptable Deviation:** +- Maximum deficit: €30 million over 3-year assessment +- Must be covered by owner equity contribution +- Aggregate break-even result monitored + +### 2.3 Squad Cost Rules (2024 onwards) + +New UEFA sustainability regulations: + +**Squad Cost Ratio:** +- Maximum 70% of relevant revenue +- Gradual implementation: + - 2023/24: 90% threshold + - 2024/25: 80% threshold + - 2025/26: 70% threshold + +**Relevant Costs:** +- Player wages +- Transfer amortization +- Agent fees + +### 2.4 Sporting Sanctions for FFP Breach + +Clubs violating FFP may face: +- Warning letters +- Fines (up to €100 million) +- Points deductions +- Transfer restrictions +- Withholding of prize money +- Squad size limitations +- Disqualification from competitions + +## Part 3: UK Governing Body Endorsement (GBE) + +### 3.1 Overview of GBE System + +Post-Brexit, international players require GBE to play in UK: + +**Applies to:** +- Premier League +- English Football League (Championship, League One, League Two) +- Scottish Premiership and leagues +- Welsh Premier League + +**Exemptions:** +- Irish citizens +- Players with UK settled status +- Existing work permits (transitional) + +### 3.2 GBE Points System + +Players earn points based on: + +**International Appearances (Maximum 15 points):** + +| FIFA Ranking | Points per Appearance | +|--------------|----------------------| +| 1-10 | 3 points | +| 11-20 | 2 points | +| 21-30 | 1.5 points | +| 31-50 | 1 point | +| 51-100 | 0.5 points | +| 101-150 | 0.25 points | +| 151+ | 0 points | + +**Percentage of Matches Required:** +- 30% of international matches over 2 years (standard) +- OR 30% in most recent year +- Youth internationals count at reduced rate + +### 3.3 Continental Club Competition Points + +**Champions League:** +- Group stage participation: +2 points +- Quarter-finals or beyond: +3 points + +**Europa League:** +- Group stage participation: +1 point +- Quarter-finals or beyond: +2 points + +**Europa Conference League:** +- Group stage: +0.5 points +- Quarter-finals or beyond: +1 point + +### 3.4 Domestic League Quality Points + +Points based on league coefficient: +- Top leagues (Premier League, La Liga, etc.): +3 points +- Second tier leagues: +2 points +- Third tier leagues: +1 point +- Lower leagues: 0 points + +### 3.5 Transfer Fee Points + +**Premier League Transfers:** +| Fee Range | Points | +|-----------|--------| +| £0-3m | 0 | +| £3-6m | 1 | +| £6-10m | 2 | +| £10-20m | 3 | +| £20-35m | 4 | +| £35m+ | 5 | + +**Championship Transfers:** +Lower thresholds apply with same points structure + +### 3.6 Exceptions Panel + +Players not meeting points threshold may apply to: + +**GBE Exceptions Panel:** +- Considers exceptional talent +- Reviews career trajectory +- Evaluates potential contribution +- Special circumstances considered + +**Factors Considered:** +- Significant transfer fee paid +- Contract value/wages offered +- Playing position scarcity +- Age and development potential +- Technical and tactical ability + +### 3.7 GBE Approval Thresholds + +**Automatic Endorsement:** +- 15+ points: Automatic approval +- 10-14 points: Conditional approval (exceptions panel review) +- Below 10 points: Exceptions panel required + +**Renewal Requirements:** +- GBE valid for initial period +- Must demonstrate playing time +- May need re-application + +## Part 4: Premier League Homegrown Player Rules + +### 4.1 Definition of Homegrown Player + +A homegrown player must have been: +- Registered with any FA or Welsh FA club +- For at least three seasons (or 36 months) +- Before the end of the season of their 21st birthday +- Regardless of nationality + +### 4.2 Squad Registration Rules + +**25-Man Squad:** +- Maximum 17 non-homegrown players +- Minimum 8 homegrown players required +- If fewer than 8 homegrown, total squad reduced + +**Under-21 Players:** +- Do not need to be registered +- Can play without counting toward 25 +- Unlimited number permitted + +### 4.3 Champions League Registration + +UEFA Club Competition requirements: +- List A: Max 25 players, similar restrictions +- List B: Unlimited under-21 players trained at club for 2 years +- Locally trained players: 8 required in List A +- 4 must be club-trained, 4 association-trained + +## Part 5: Agent Regulations and Compliance + +### 5.1 FIFA Football Agent Regulations (FFAR) + +Since January 2023: + +**Agent Licensing:** +- Must pass FIFA agent exam +- License required for all transactions +- Continuing education requirements +- Background checks mandatory + +**Fee Caps:** +- Maximum 5% of player contract value +- Maximum 10% of transfer fee (for selling club) +- Maximum 3% of transfer fee (for buying club) +- Dual representation prohibited + +### 5.2 Agent Commission Rules + +**Permitted:** +- Transparent commission agreements +- Single client representation per transaction +- Written contracts with clear terms +- Payment through approved channels + +**Prohibited:** +- Undisclosed payments +- Representing multiple parties +- Payments to minors +- Ownership of player economic rights + +### 5.3 Due Diligence Requirements + +Clubs must verify: +- Agent license validity +- No conflicts of interest +- Proper contractual authority +- Compliance with fee caps +- Transparent payment structures + +## Part 6: Anti-Discrimination and Integrity + +### 6.1 FIFA Anti-Discrimination Standards + +All transfers must comply with: +- Equal treatment principles +- No discrimination based on nationality, race, religion +- Protection of player rights +- Fair labor practices + +### 6.2 Match-Fixing Prevention + +Recruitment must consider: +- Background checks for integrity concerns +- No association with match-fixing +- Cooperation with integrity units +- Reporting obligations + +### 6.3 Doping Compliance + +Players must be: +- Registered in ADAMS (Anti-Doping Administration & Management System) +- Available for out-of-competition testing +- No prior doping violations (or completed sanctions) +- Subject to whereabouts requirements + +## Part 7: Work Permit and Visa Requirements + +### 7.1 UK Work Permit Categories + +**Sportsperson Visa (T5):** +- For elite athletes and coaches +- GBE required for footballers +- Valid for up to 12 months initially +- Renewable based on continued employment + +**Skilled Worker Visa:** +- Alternative route for some sports +- Requires sponsorship from licensed employer +- Points-based assessment +- Minimum salary thresholds apply + +### 7.2 EU/EEA Player Changes Post-Brexit + +Since January 2021: +- No automatic right to work in UK +- GBE system replaces free movement +- Existing EU players with settled status exempt +- New EU signings need GBE approval + +### 7.3 Processing Timelines + +**GBE Application:** +- Standard: 2-3 weeks +- Exceptions panel: Additional 2-4 weeks + +**Visa Application:** +- Priority: 5 working days +- Standard: 3-8 weeks +- Delays possible during peak periods + +## Part 8: Contract and Registration Requirements + +### 8.1 Standard Contract Terms + +**FIFA Requirements:** +- Minimum contract length: Season (amateur) to 5 years (professional) +- Clear termination clauses +- Salary and bonus structures defined +- Image rights provisions +- Compliance with local labor law + +### 8.2 Registration Deadlines + +**Premier League:** +- Summer window: Mid-June to end of August +- Winter window: January (approximately 4 weeks) +- Emergency loan windows: Goalkeepers only, specific periods + +**UEFA Competitions:** +- List A: Registered by deadline before group stage +- List B: Can be updated through season +- Mid-season changes limited + +### 8.3 Loan Agreements + +**FIFA Loan Regulations:** +- Maximum 6 loans in (domestic) +- Maximum 6 loans out (domestic) +- Loan fees must be disclosed +- Recall clauses regulated +- Option/obligation to buy governed + +## Part 9: Compliance Checklist + +### 9.1 Pre-Transfer Due Diligence + +Before completing a transfer: + +☐ Verify player identity and age documents +☐ Check current contract status +☐ Confirm no ongoing disputes +☐ Review disciplinary history +☐ Conduct medical examination +☐ Verify work permit/GBE eligibility +☐ Check third-party ownership status +☐ Confirm agent authorization +☐ Review training compensation obligations +☐ Calculate solidarity contributions + +### 9.2 Transfer Completion Requirements + +To finalize transfer: + +☐ Agreement between clubs +☐ Player contract signed +☐ ITC requested and received +☐ Registration submitted to FA +☐ Work permit/visa obtained +☐ Medical clearance +☐ Payment terms agreed +☐ FIFA TMS submission + +### 9.3 Post-Transfer Compliance + +After transfer completion: + +☐ Register in domestic competition +☐ UEFA registration (if applicable) +☐ ADAMS registration update +☐ Report to tax authorities +☐ Image rights agreements finalized +☐ Insurance coverage confirmed +☐ Solidarity payments processed +☐ Agent fees paid and reported + +## Conclusion + +Elite athlete recruitment in football requires compliance with multiple regulatory frameworks: + +1. **FIFA Regulations**: Govern international transfers, protect minors, ensure training compensation +2. **UEFA FFP**: Maintain financial stability and competitive balance +3. **UK GBE System**: Post-Brexit work authorization for international players +4. **Agent Regulations**: Transparent and ethical representation +5. **Integrity Standards**: Anti-discrimination, anti-doping, match-fixing prevention + +Successful recruitment requires thorough understanding of these interconnected regulations and meticulous compliance processes. + +--- + +*Document Version: 2024.1* +*Last Updated: January 2024* +*This guide is for informational purposes. Always verify current requirements with governing bodies.* diff --git a/graphrag-ollama-config/input/internationa.txt b/graphrag-ollama-config/input/internationa.txt new file mode 100644 index 0000000..3f33279 --- /dev/null +++ b/graphrag-ollama-config/input/internationa.txt @@ -0,0 +1,115 @@ +Dashboard Title: STUDENT VISA & ADMISSION ELIGIBILITY ANALYTICS (DETAILED) Total Candidates Evaluated: 7,773 + +1. ACADEMIC ELIGIBILITY METRICS Minimum GPA Requirement (Region specific): + +High Eligibility (> 3.5 GPA): 2,597 candidates (Ivy League / Top Tier Eligible) + +Standard Eligibility (3.0 - 3.49 GPA): 3,115 candidates (State/Public Universities) + +Conditional Eligibility (2.5 - 2.99 GPA): 1,544 candidates (Pathway Programs/Community Colleges) + +At-Risk/Ineligible (< 2.5 GPA): 517 candidates + +Standardized Test Status (GRE/GMAT/SAT): + +Scores Verified (Above Threshold): 3,200 candidates + +Waiver Granted (Experience/GPA based): 1,850 candidates + +Pending Scores: 2,123 candidates + +Below Threshold (Retake Required): 600 candidates + +2. LANGUAGE PROFICIENCY CRITERIA (CEFR Standards) English (IELTS / TOEFL / PTE): + +C1/C2 Advanced (IELTS 7.5+ / TOEFL 100+): 1,940 candidates + +B2 Upper Intermediate (IELTS 6.5 / TOEFL 80-99): 4,200 candidates (Meets standard US/UK requirement) + +B1 Intermediate (IELTS 5.5 - 6.0): 1,250 candidates (Requires Pre-sessional English) + +Below Requirement: 383 candidates + +European Languages (German/French - For Public Universities): + +B2/C1 Certified (Direct Entry): 450 candidates + +A1/A2 Basic (Conditional Admission): 890 candidates + +No Proficiency (English-taught programs only): 6,433 candidates + +3. FINANCIAL ELIGIBILITY & FUNDING PROOF USA (Liquid Assets ~ $35,000 - $60,000/year): + +Fully Funded (Scholarship/Assistantship): 850 candidates + +Self-Funded (Bank Statement Verified): 2,400 candidates + +Loan Sanctioned: 1,200 candidates + +Insufficient Funds (Risk of 214(b) Denial): 450 candidates + +Europe/Germany (Blocked Account ~ €11,208/year): + +Deposit Confirmed: 1,800 candidates + +Sponsorship Letter Verified: 600 candidates + +Pending Financial Proof: 473 candidates + +4. VISA COMPLIANCE & TIES TO HOME COUNTRY Gap Year Justification: + +No Gap (Fresh Graduate): 4,500 candidates + +Gap Explained (Work Experience verified): 2,100 candidates + +Unexplained Gap (> 1 year): 673 candidates (High Visa Risk) + +Immigration Intent (USA Specific): + +Clear Non-Immigrant Intent Demonstrated: 5,200 candidates + +Potential Dual Intent Flagged: 800 candidates + +Previous Visa Refusals: 320 candidates (Requires complex review) + +5. REGION-SPECIFIC ELIGIBILITY CHECKS United States (F-1 Criteria): + +I-20 Issued: 2,100 + +DS-160 Completed: 1,850 + +SEVIS Fee Paid: 1,900 + +United Kingdom (CAS Criteria): + +CAS Letter Received: 950 + +Tuberculosis (TB) Test Cleared: 900 + +Credibility Interview Passed: 880 + +Schengen Area (Germany/France): + +APS Certificate (India/China specific): 650 Verified + +Campus France/Uni-Assist Evaluation Cleared: 820 + +Reasons for Ineligibility (Top Factors): + +Academic Mismatch: 40% (Background does not match Master's prereqs) + +Financial Shortfall: 25% (Cannot show liquid funds for Year 1) + +Language Barrier: 20% (Failed to meet IELTS 6.5 minimum) + +Unjustified Study Gap: 10% + +Age/Visa History: 5% + +Metadata: + +Compliance Standard: US INA Act 214(b) / UKVI Tier 4 Guidelines + +Last Updated: 2025-04-14 + +Verification Level: Pre-Screening Phase 2 \ No newline at end of file diff --git a/graphrag-ollama-config/input/ncaa_eligibility_guide.txt b/graphrag-ollama-config/input/ncaa_eligibility_guide.txt new file mode 100644 index 0000000..2c4f90c --- /dev/null +++ b/graphrag-ollama-config/input/ncaa_eligibility_guide.txt @@ -0,0 +1,367 @@ +# NCAA Eligibility Requirements for Student Athletes + +## Executive Summary + +The National Collegiate Athletic Association (NCAA) establishes eligibility requirements for student-athletes participating in Division I, Division II, and Division III athletics. These requirements ensure that student-athletes maintain academic standards while competing at the collegiate level. + +## Part 1: Initial Eligibility Requirements + +### 1.1 NCAA Eligibility Center Registration + +All prospective student-athletes must register with the NCAA Eligibility Center (formerly NCAA Clearinghouse). This registration process includes: + +- Creating an account at eligibilitycenter.org +- Submitting official high school transcripts +- Submitting ACT or SAT test scores +- Paying the registration fee ($90 for domestic students) +- Requesting amateurism certification + +### 1.2 Academic Requirements for Division I + +**Core Course Requirements:** +- 16 core courses required for full qualifier status +- 4 years of English +- 3 years of mathematics (Algebra I or higher) +- 2 years of natural/physical science (including 1 lab course) +- 1 additional year of English, math, or natural science +- 2 years of social science +- 4 additional years from any core area + +**GPA and Test Score Requirements:** +- Minimum core GPA: 2.3 for full qualifier +- Sliding scale: Higher GPA allows lower test scores +- SAT minimum: 400 combined (math and critical reading) +- ACT sum score minimum: 37 + +**Sliding Scale Example (Division I):** +| Core GPA | SAT Score | ACT Sum Score | +|----------|-----------|---------------| +| 2.300 | 1010 | 86 | +| 2.400 | 960 | 82 | +| 2.500 | 910 | 78 | +| 2.600 | 860 | 74 | +| 2.700 | 820 | 70 | +| 2.800 | 780 | 66 | +| 2.900 | 740 | 62 | +| 3.000 | 700 | 58 | +| 3.200 | 620 | 52 | +| 3.550+ | 400 | 37 | + +### 1.3 Academic Requirements for Division II + +**Core Course Requirements:** +- 16 core courses required +- 3 years of English +- 2 years of mathematics (Algebra I or higher) +- 2 years of natural/physical science (including 1 lab course) +- 3 additional years of English, math, or science +- 2 years of social science +- 4 additional years from any core area + +**GPA and Test Score Requirements:** +- Minimum core GPA: 2.2 +- Sliding scale similar to Division I +- SAT minimum: 400 combined +- ACT sum score minimum: 37 + +### 1.4 Division III Requirements + +Division III does not require NCAA Eligibility Center certification. However, student-athletes must: +- Be admitted to the institution as a regular student +- Meet institutional academic standards +- Maintain satisfactory academic progress + +## Part 2: Continuing Eligibility Requirements + +### 2.1 Progress Toward Degree (PTD) + +Student-athletes must demonstrate progress toward a degree: + +**Division I:** +- Complete 40% of degree requirements by end of Year 2 +- Complete 60% of degree requirements by end of Year 3 +- Complete 80% of degree requirements by end of Year 4 + +**Division II:** +- Slightly less stringent requirements +- Must maintain progress toward graduation + +### 2.2 Academic Performance Requirements + +**Academic Progress Rate (APR):** +- Measures eligibility, retention, and graduation +- Maximum points: 1,000 (all athletes eligible and retained) +- Minimum team APR: 930 (2024-25 season) +- Penalties for teams below threshold + +**Graduation Success Rate (GSR):** +- Tracks graduation rates of scholarship athletes +- Accounts for transfers and mid-year enrollees +- Federal graduation rate comparison metric + +### 2.3 Credit Hour Requirements + +**Division I:** +- Minimum 6 credit hours per term to compete +- 18 credit hours earned since previous fall term +- 24 credit hours earned since start of previous academic year +- Must be enrolled full-time (12+ credit hours) + +**Division II:** +- Similar credit hour requirements +- Must maintain full-time enrollment +- Satisfactory progress toward degree + +## Part 3: Amateurism Requirements + +### 3.1 Definition of Amateurism + +Student-athletes must maintain amateur status by: +- Not signing contracts with professional teams +- Not receiving salary for sport participation +- Not competing with professionals (with exceptions) +- Not accepting prize money above expenses + +### 3.2 Permissible Activities + +Student-athletes MAY: +- Receive athletics scholarships +- Accept limited expense reimbursement for tryouts +- Participate in Olympic or national team competitions +- Receive awards (subject to value limits) + +### 3.3 Prohibited Activities + +Student-athletes may NOT: +- Receive payment for their athletic abilities +- Sign with agents (affects eligibility) +- Use athletic skill for pay (endorsements - pre-NIL era) +- Accept excessive gifts or benefits + +### 3.4 Name, Image, and Likeness (NIL) Rules + +Since July 2021, NCAA student-athletes may: +- Profit from their name, image, and likeness +- Sign endorsement deals +- Create and monetize social media content +- Participate in promotional activities + +NIL Guidelines: +- Cannot be used as recruiting inducement +- Must comply with state NIL laws +- Cannot conflict with team or school contracts +- Must disclose NIL activities to institution + +## Part 4: Transfer Eligibility + +### 4.1 Transfer Portal + +The NCAA Transfer Portal allows: +- Athletes to notify their intent to transfer +- Coaches from other schools to contact them +- Streamlined transfer process + +### 4.2 Transfer Requirements + +**One-Time Transfer Exception:** +- Athletes in good academic standing +- May transfer and compete immediately +- Applies to most sports (football has limitations) + +**Notification Requirements:** +- Must enter transfer portal by specific dates +- Spring sports: May 1 deadline +- Fall sports: May 1 deadline (varies by sport) +- Winter sports: May 1 deadline + +### 4.3 Academic Requirements for Transfers + +Transferring athletes must: +- Be academically eligible at previous institution +- Meet minimum credit hour requirements +- Have transferable credits toward degree +- Meet receiving institution's admission standards + +## Part 5: Medical and Health Requirements + +### 5.1 Pre-Participation Medical Examination + +All student-athletes must complete: +- Comprehensive medical history +- Physical examination +- Cardiac screening (ECG for high-risk sports) +- Sickle cell trait testing + +### 5.2 Concussion Protocol + +NCAA concussion guidelines require: +- Baseline testing before competition +- Immediate removal from play if concussion suspected +- Gradual return-to-play protocol +- Medical clearance before return + +### 5.3 Drug Testing + +Student-athletes are subject to: +- NCAA year-round drug testing +- Championship testing +- Institutional drug testing programs +- Penalties for positive tests (up to permanent ban) + +Banned Substances Include: +- Performance enhancing drugs +- Stimulants +- Anabolic agents +- Certain hormones +- Recreational drugs + +## Part 6: Financial Aid and Scholarships + +### 6.1 Types of Athletics Aid + +**Full Scholarship (Grant-in-Aid):** +- Tuition and fees +- Room and board +- Required course books +- Can include cost of attendance stipend + +**Partial Scholarship:** +- Portion of full scholarship +- Common in equivalency sports +- Can be combined with academic aid + +### 6.2 Scholarship Limitations by Sport + +**Head Count Sports (Full scholarships only):** +- Football (Division I-A): 85 scholarships +- Men's Basketball: 13 scholarships +- Women's Basketball: 15 scholarships +- Women's Volleyball: 12 scholarships +- Women's Tennis: 8 scholarships +- Women's Gymnastics: 12 scholarships + +**Equivalency Sports (Can divide scholarships):** +- Baseball: 11.7 equivalencies +- Soccer: 9.9 equivalencies +- Track and Field: 12.6 equivalencies + +### 6.3 Academic Scholarships + +Student-athletes may also receive: +- Merit-based academic scholarships +- Need-based financial aid +- Institutional grants +- External scholarships (with limitations) + +## Part 7: Compliance and Violations + +### 7.1 Major Violations + +Major violations include: +- Academic fraud +- Impermissible benefits to recruits +- Booster involvement in recruitment +- Falsifying eligibility documents +- Systematic failures to comply with rules + +Penalties for Major Violations: +- Scholarship reductions +- Postseason bans +- Show-cause orders for coaches +- Vacated wins +- Probation periods + +### 7.2 Secondary Violations + +Secondary violations are: +- Isolated or inadvertent +- Provide limited recruiting advantage +- Minor extra benefits + +Typical Secondary Violations: +- Recruiting contact on wrong date +- Providing small impermissible gift +- Practice outside allowed hours + +### 7.3 Self-Reporting + +Institutions must: +- Self-report violations to NCAA +- Implement corrective actions +- Cooperate with investigations +- Maintain compliance programs + +## Part 8: International Student-Athlete Requirements + +### 8.1 Academic Evaluation + +International students must: +- Submit international transcripts for evaluation +- Meet core course requirements (evaluated differently) +- Demonstrate English proficiency +- Complete credential evaluation through approved service + +### 8.2 Amateurism Certification + +International students must verify: +- No professional contracts +- No payment for athletic participation +- Limited prize money received +- Proper documentation of all competitions + +### 8.3 Visa and Immigration Compliance + +International student-athletes must maintain: +- Valid F-1 or J-1 visa status +- Full-time enrollment +- Work authorization for compensation +- Compliance with immigration regulations + +## Part 9: Academic Support Services + +### 9.1 Required Academic Support + +Institutions must provide: +- Academic advisors for student-athletes +- Tutoring services +- Study hall requirements (for certain athletes) +- Priority registration (permitted) + +### 9.2 Freshman Requirements + +First-year student-athletes often have: +- Mandatory study hall hours (typically 8-10 hours/week) +- Academic monitoring +- Progress reports from professors +- Early intervention programs + +### 9.3 Success Programs + +Academic success programs include: +- Summer bridge programs +- Learning specialists +- Career counseling +- Life skills development + +## Part 10: Conclusion + +NCAA eligibility is a complex system designed to: +- Ensure academic integrity +- Maintain competitive balance +- Protect student-athlete welfare +- Preserve the amateur model (evolving with NIL) + +Student-athletes must understand and comply with: +- Initial eligibility requirements +- Continuing eligibility standards +- Amateurism rules +- Transfer regulations +- Financial aid limitations + +Success as a student-athlete requires balancing athletic and academic excellence while navigating NCAA regulations. + +--- + +*Document Version: 2024.2* +*Last Updated: January 2024* +*This guide is for informational purposes. Always verify current requirements with NCAA.org.* diff --git a/graphrag-ollama-config/input/sports.txt b/graphrag-ollama-config/input/sports.txt new file mode 100644 index 0000000..df412c0 --- /dev/null +++ b/graphrag-ollama-config/input/sports.txt @@ -0,0 +1,127 @@ +Dashboard Title: ELITE ATHLETE RECRUITMENT & ELIGIBILITY DASHBOARD Total Athletes Scouted: 5,432 + +1. ATHLETIC PERFORMANCE & PHYSICAL STANDARDS Performance Tiers (Sport Specific): + +Tier 1 (World Class/Elite): 450 athletes (Olympic Qualifier / Top 5 League Starter) + +Tier 2 (Professional/National Level): 1,890 athletes (First Division Regulars) + +Tier 3 (Developmental/Academy): 2,150 athletes (High Potential U-21) + +Tier 4 (Amateur/Recreational): 942 athletes (Club Level) + +Physical Metrics (Verified): + +Speed/Agility (Top 10%): 850 athletes (Sub-11s 100m / High V02 Max) + +Strength/Power: 1,200 athletes (Meets Combine Benchmarks) + +Technical Proficiency Score (>8/10): 1,500 athletes + +Injury Prone / Medical Flag: 320 athletes (Failed Medical Screening) + +2. LEAGUE & REGULATORY COMPLIANCE (Visa & Work Permit) United Kingdom (GBE - Governing Body Endorsement): + +Auto-Pass (15+ Points): 350 athletes (Regular International Caps for Top 50 Nations) + +Exceptions Panel (10-14 Points): 120 athletes (Requires Appeal) + +Ineligible (<10 Points): 400 athletes (Cannot obtain Sportsperson Visa) + +United States (P-1 Visa / NCAA Eligibility): + +P-1A (Internationally Recognized): 280 athletes + +NCAA Division I Qualifier: 1,400 athletes (Amateur Status Verified) + +NCAA Academic Non-Qualifier: 150 athletes (GPA < 2.3) + +European Union (Non-EU Quota Rules): + +EU Passport Holders (No Restrictions): 1,200 athletes + +Non-EU (Quota Slot Required): 850 athletes + +Kolpak/Cotonou Eligible: 300 athletes + +3. SCOUTING & RECRUITMENT STATUS Current Pipeline Stage: + +Initial Observation: 2,100 athletes + +Detailed Data Analysis: 1,500 athletes + +In-Person Scouting: 900 athletes + +Trial Invited: 450 athletes + +Contract Offer Extended: 182 athletes + +Signed: 45 athletes + +4. ACADEMIC ELIGIBILITY (For Student-Athletes/US Scholarships) GPA Requirements: + +High Academic (GPA > 3.5): 950 athletes (Ivy League/Stanford Eligible) + +NCAA Standard (GPA 2.3 - 3.0): 1,800 athletes + +Junior College Route (GPA < 2.3): 450 athletes + +Standardized Tests (SAT/ACT): + +Scores Submitted & Verified: 1,200 athletes + +Test Optional/Waiver: 800 athletes + +Pending Scores: 600 athletes + +5. FINANCIAL VALUATION & BUDGET FIT Market Value Estimation: + +Premium (> $10M Transfer Fee): 25 athletes + +High Value ($1M - $10M): 150 athletes + +Mid-Range ($100k - $1M): 800 athletes + +Free Agent / Scholarship (Zero Fee): 4,457 athletes + +Funding Status: + +Full Scholarship Available: 300 spots + +Partial Scholarship: 500 spots + +Club Budget Approved: $15.5M Total + +Most Targeted Sports by Region: + +Football (Soccer): 2,100 athletes (Primary: South America, Europe, West Africa) + +Basketball: 1,200 athletes (Primary: USA, Balkans, Spain) + +Track & Field: 800 athletes (Primary: USA, Jamaica, East Africa) + +Cricket: 650 athletes (Primary: India, Australia, UK) + +Rugby: 350 athletes (Primary: New Zealand, South Africa, France) + +Reasons for Ineligibility/Rejection (Top Factors): + +Visa Denial: 30% (Failed to meet GBE points or "Internationally Recognized" standard) + +Medical Failure: 25% (Undisclosed chronic injuries) + +Academic Ineligibility: 20% (Low GPA/SAT for NCAA) + +Character/Behavioral Flag: 10% + +Agent/Representation Issues: 15% + +Metadata: + +Compliance Standard: FIFA Transfer Regulations / NCAA Bylaws / UK Home Office + +Last Updated: 2025-04-15 09:00:00 UTC + +Data Source: Global Scouting Network (Wyscout, Hudl, NCAA Clearinghouse) + +Version: 3.0.1 \ No newline at end of file diff --git a/graphrag-ollama-config/input/student_visa_guide.txt b/graphrag-ollama-config/input/student_visa_guide.txt new file mode 100644 index 0000000..9cc1aaf --- /dev/null +++ b/graphrag-ollama-config/input/student_visa_guide.txt @@ -0,0 +1,344 @@ +# Student Visa Eligibility Requirements: A Comprehensive Guide + +## Chapter 1: Introduction to Student Visas + +Student visas are non-immigrant visas that allow foreign nationals to enter a country for the purpose of pursuing education at accredited institutions. The requirements for obtaining a student visa vary significantly between countries, but all share common elements designed to ensure that applicants are genuine students with the financial means to support themselves during their studies. + +### 1.1 Purpose of Student Visas + +The primary purpose of student visas is to: +- Facilitate international education exchange +- Ensure students have legitimate educational objectives +- Verify financial capability to cover tuition and living expenses +- Maintain immigration control while promoting educational opportunities +- Foster cultural exchange and international understanding + +## Chapter 2: USA F-1 Student Visa Requirements + +### 2.1 Overview of the F-1 Visa + +The F-1 visa is the most common type of student visa for the United States. It is issued to students who wish to attend academic programs at accredited U.S. institutions, including universities, colleges, high schools, private elementary schools, seminaries, conservatories, and language training programs. + +### 2.2 Eligibility Requirements for F-1 Visa + +To qualify for an F-1 student visa, applicants must meet the following requirements: + +**Academic Requirements:** +- Acceptance letter from a SEVP-certified school (Student and Exchange Visitor Program) +- Form I-20 issued by the designated school official (DSO) +- Proof of English language proficiency (TOEFL score minimum 80 iBT, IELTS minimum 6.5) +- Academic transcripts from previous education +- Standardized test scores (SAT, GRE, or GMAT as applicable) + +**Financial Requirements:** +- Bank statements showing sufficient funds to cover first year of study +- Minimum financial proof: $50,000 - $70,000 for one academic year +- Sponsorship letters if funded by parents, relatives, or organizations +- Scholarship award letters if applicable +- Affidavit of support from financial sponsors + +**Documentation Requirements:** +- Valid passport with at least 6 months validity beyond intended stay +- DS-160 Online Visa Application form +- Visa application fee payment receipt ($185 USD) +- SEVIS fee payment receipt ($350 USD) +- Passport-sized photographs meeting U.S. specifications +- Proof of ties to home country (property, employment, family) + +### 2.3 F-1 Visa Interview Process + +The visa interview is a critical component of the F-1 application: + +1. **Schedule Interview**: Book appointment at U.S. Embassy or Consulate +2. **Prepare Documents**: Organize all required documentation +3. **Interview Questions**: Be prepared to discuss: + - Why you chose this particular school + - Your intended major and career plans + - How you will finance your education + - Your plans to return to your home country after graduation +4. **Biometrics**: Fingerprint scanning during the interview + +### 2.4 F-1 Visa Denial Reasons + +Common reasons for F-1 visa denial include: +- Insufficient proof of financial support +- Weak ties to home country +- Incomplete or inconsistent documentation +- Poor interview performance +- Previous immigration violations +- Inadequate English language skills +- Unclear educational or career objectives + +## Chapter 3: UK Tier 4 (Student) Visa Requirements + +### 3.1 Overview of UK Student Visa + +The UK Student Visa (formerly Tier 4) allows international students to study at licensed institutions in the United Kingdom. The visa system uses a points-based assessment where applicants must score 70 points to qualify. + +### 3.2 Points-Based Requirements + +**Confirmation of Acceptance for Studies (CAS) - 50 Points:** +- Valid CAS reference number from a licensed sponsor +- CAS must be issued within 6 months of application +- Course must be at RQF Level 3 or above (for general student visa) + +**Financial Requirements - 20 Points:** +- Tuition fees: Amount stated on CAS +- Living costs (maintenance): + - London: £1,334 per month (up to 9 months = £12,006) + - Outside London: £1,023 per month (up to 9 months = £9,207) +- Funds must be held for at least 28 consecutive days +- Official financial documents required + +### 3.3 Documentation for UK Student Visa + +Required documents include: +- Valid passport or travel document +- CAS reference number and supporting documents +- Proof of English language ability (IELTS UKVI minimum 5.5 overall) +- Financial evidence (bank statements, loan letters, scholarship letters) +- Tuberculosis test results (if from listed countries) +- Academic Technology Approval Scheme (ATAS) certificate (for certain subjects) +- Parental consent (if under 18) + +### 3.4 English Language Requirements + +Applicants must prove English proficiency through: +- IELTS for UKVI: Minimum 5.5 overall for degree level courses +- IELTS for UKVI: Minimum 4.0 for below degree level courses +- Trinity ISE: Accepted at various levels +- Pearson PTE Academic UKVI +- Cambridge English qualifications +- Being a national of a majority English-speaking country +- Having completed a qualification taught in English + +### 3.5 UK Student Visa Fees and Timeline + +- Visa application fee: £490 (outside UK) +- Immigration Health Surcharge: £776 per year +- Processing time: 3-4 weeks standard +- Priority processing available for additional fee +- Super Priority: Decision within 24 hours (limited availability) + +## Chapter 4: Schengen Student Visa Requirements + +### 4.1 Overview of Schengen Student Visa + +The Schengen student visa allows students to study in any of the 27 Schengen Area countries. Each country has specific requirements, but the general framework is consistent. + +### 4.2 Common Schengen Requirements + +**Academic Requirements:** +- Acceptance letter from recognized educational institution +- Proof of enrollment or pre-enrollment +- Academic transcripts and diplomas +- Language proficiency (varies by country and program language) + +**Financial Requirements:** +- Proof of sufficient funds for duration of stay +- Germany: €934 per month in blocked account (€11,208 annually) +- France: €615 per month minimum +- Netherlands: €950 per month +- Bank guarantee or scholarship proof + +**Insurance Requirements:** +- Travel health insurance with minimum €30,000 coverage +- Coverage for medical emergencies and repatriation +- Valid for entire Schengen Area + +### 4.3 Country-Specific Requirements + +**Germany:** +- Blocked account (Sperrkonto) with €11,208 minimum +- Health insurance (public or private) +- Proof of accommodation +- Uni-Assist application for most universities + +**France:** +- Campus France registration and interview +- Proof of accommodation +- Birth certificate (translated and apostilled) +- Long-stay visa (VLS-TS) application + +**Netherlands:** +- MVV (Entry Visa) for stays over 90 days +- Proof of sufficient funds (€950/month) +- Health insurance +- Residence permit application + +## Chapter 5: Financial Requirements Comparison + +### 5.1 Summary of Financial Requirements by Country + +| Country | Annual Amount Required | Currency | Notes | +|---------|----------------------|----------|-------| +| USA (F-1) | $50,000 - $70,000 | USD | Varies by institution | +| UK | £12,006 - £15,000+ | GBP | London vs outside London | +| Germany | €11,208 | EUR | Blocked account required | +| France | €7,380 | EUR | €615 per month | +| Netherlands | €11,400 | EUR | €950 per month | +| Canada | CAD $20,635 | CAD | Plus tuition | +| Australia | AUD $24,505 | AUD | Plus tuition and travel | + +### 5.2 Acceptable Financial Documentation + +- Bank statements (3-6 months history) +- Fixed deposit certificates +- Scholarship award letters +- Education loan approval letters +- Sponsor's financial documents with affidavit +- Government-issued financial guarantees +- Employer-sponsored letters with salary details + +## Chapter 6: Common Rejection Reasons and How to Avoid Them + +### 6.1 Top 10 Visa Rejection Reasons + +1. **Insufficient Financial Documentation** + - Solution: Provide comprehensive bank statements, show consistent balance + +2. **Weak Ties to Home Country** + - Solution: Document property ownership, family ties, job offers upon return + +3. **Incomplete Application** + - Solution: Use checklist, double-check all documents before submission + +4. **Poor Interview Performance** + - Solution: Practice common questions, be confident and honest + +5. **Inconsistent Information** + - Solution: Ensure all documents align with application statements + +6. **Inadequate English Proficiency** + - Solution: Achieve required test scores before applying + +7. **Unclear Study Plans** + - Solution: Articulate clear academic and career goals + +8. **Previous Immigration Violations** + - Solution: Address past issues honestly, provide explanations + +9. **Health or Security Concerns** + - Solution: Complete medical exams, provide police clearances + +10. **Fraudulent Documents** + - Solution: Only submit genuine, verifiable documents + +### 6.2 Visa Interview Best Practices + +- Arrive early (at least 30 minutes before appointment) +- Dress professionally and appropriately +- Bring organized documents in a clear folder +- Answer questions directly and concisely +- Maintain eye contact and confident body language +- Be prepared to explain gaps in education or employment +- Have clear reasons for choosing the specific institution +- Demonstrate knowledge about your program of study + +## Chapter 7: Post-Visa Approval Requirements + +### 7.1 Pre-Departure Checklist + +After visa approval, students must: +- Book flights and arrange airport pickup +- Arrange accommodation (on-campus or off-campus) +- Complete health insurance enrollment +- Attend pre-departure orientation (if offered) +- Carry all original documents during travel +- Have emergency contacts and embassy information +- Exchange currency and set up international banking + +### 7.2 Arrival and Registration Requirements + +**USA:** +- Report to designated school within 30 days of program start +- Complete SEVIS registration +- Attend mandatory orientation +- Obtain Social Security Number (if eligible for work) + +**UK:** +- Collect Biometric Residence Permit (BRP) within 10 days +- Register with local police (if required) +- Open UK bank account +- Register with National Health Service (NHS) + +**Schengen Countries:** +- Register with local authorities (varies by country) +- Apply for residence permit (if required) +- Register at educational institution +- Set up local bank account and health insurance + +## Chapter 8: Work Rights for International Students + +### 8.1 USA F-1 Work Authorization + +- On-campus employment: Up to 20 hours/week during term +- Curricular Practical Training (CPT): Work related to field of study +- Optional Practical Training (OPT): 12 months post-graduation +- STEM OPT Extension: Additional 24 months for STEM graduates + +### 8.2 UK Student Visa Work Rights + +- During term: 20 hours/week for degree-level students +- During vacations: Full-time work permitted +- Post-Study Work Visa: 2-3 years depending on qualification + +### 8.3 Schengen Work Rights (varies by country) + +**Germany:** +- 120 full days or 240 half days per year +- Unlimited work for student jobs at university + +**France:** +- 964 hours per year (approximately 20 hours/week) +- Work authorization automatic with student visa + +**Netherlands:** +- 16 hours/week during academic year +- Full-time during summer months +- Work permit (TWV) required + +## Chapter 9: Visa Extensions and Status Changes + +### 9.1 Extending Student Visa Status + +Students may need to extend their visa if: +- Program takes longer than expected +- Transferring to a new program +- Starting additional degree programs +- Medical or personal emergencies + +Requirements for extension: +- Maintain valid student status +- Continue full-time enrollment +- Maintain satisfactory academic progress +- Apply before current status expires +- Provide updated financial documentation + +### 9.2 Changing Student Status + +Options for status change include: +- Student to work visa (post-graduation) +- Student to dependent status +- Student to permanent resident (limited paths) +- Transfer between institutions (same visa category) + +## Chapter 10: Conclusion + +Obtaining a student visa requires careful preparation, comprehensive documentation, and clear communication of educational objectives. Success rates improve significantly when applicants: + +1. Start preparation at least 6 months before intended travel +2. Gather all required documents systematically +3. Demonstrate genuine student intent +4. Show strong ties to home country +5. Provide clear financial evidence +6. Practice for visa interviews +7. Follow all application guidelines precisely + +The investment in proper visa preparation pays dividends in achieving educational goals abroad. Students who approach the process methodically and honestly have the highest success rates in obtaining their student visas. + +--- + +*Document Version: 2024.1* +*Last Updated: January 2024* +*This guide is for informational purposes only. Always verify current requirements with official government sources.* diff --git "a/graphrag-ollama-config/input/text_TH\303\224NG_T\306\257_S\341\273\254A_\304\220\341\273\224I_B\341\273\224_SUNG_M\341\273\230T_S\341\273\220_\304\220I\341\273\200U_C\341\273\246A_TH\303\224NG_T\306\257__85df99d6.txt" "b/graphrag-ollama-config/input/text_TH\303\224NG_T\306\257_S\341\273\254A_\304\220\341\273\224I_B\341\273\224_SUNG_M\341\273\230T_S\341\273\220_\304\220I\341\273\200U_C\341\273\246A_TH\303\224NG_T\306\257__85df99d6.txt" new file mode 100644 index 0000000..38d2356 --- /dev/null +++ "b/graphrag-ollama-config/input/text_TH\303\224NG_T\306\257_S\341\273\254A_\304\220\341\273\224I_B\341\273\224_SUNG_M\341\273\230T_S\341\273\220_\304\220I\341\273\200U_C\341\273\246A_TH\303\224NG_T\306\257__85df99d6.txt" @@ -0,0 +1,314 @@ +# Source: DIRECT TEXT INPUT +# Title: THÔNG TƯ +SỬA ĐỔI, BỔ SUNG MỘT SỐ ĐIỀU CỦA THÔNG TƯ SỐ 73/2021/TT-BCA NGÀY 29/6/2021 CỦA BỘ TRƯỞNG BỘ CÔNG AN QUY ĐỊNH VỀ MẪU HỘ CHIẾU, GIẤY THÔNG HÀNH VÀ CÁC BIỂU MẪU LIÊN QUAN +# Ingested: 2026-01-30 20:22:28 +# Word Count: 2896 +# Character Count: 15650 + +--- +BỘ CÔNG AN +___________ CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +_________________________ +Số: 68/2022/TT-BCA Hà Nội, ngày 31 tháng 12 năm 2022 + + +THÔNG TƯ +SỬA ĐỔI, BỔ SUNG MỘT SỐ ĐIỀU CỦA THÔNG TƯ SỐ 73/2021/TT-BCA NGÀY 29/6/2021 CỦA BỘ TRƯỞNG BỘ CÔNG AN QUY ĐỊNH VỀ MẪU HỘ CHIẾU, GIẤY THÔNG HÀNH VÀ CÁC BIỂU MẪU LIÊN QUAN +______________ + +Căn cứ Nghị quyết số 76/2022/QH15 ngày 15/11/2022 của Quốc hội về Kỳ họp thứ 4, Quốc hội khóa XV; +Căn cứ Luật xuất cảnh, nhập cảnh của công dân Việt Nam ngày 22 tháng 11 năm 2019; +Căn cứ Nghị định số 01/2018/NĐ-CP ngày 06 tháng 8 năm 2018 của Chính phủ quy định chức năng, nhiệm vụ, quyền hạn và cơ cấu tổ chức của Bộ Công an; +Theo đề nghị của Cục trưởng Cục Quản lý xuất nhập cảnh; +Bộ trưởng Bộ Công an ban hành Thông tư sửa đổi, bổ sung một số điều của Thông tư số 73/2021/TT-BCA ngày 29/6/2021 quy định về mẫu hộ chiếu, giấy thông hành và các biểu mẫu liên quan. +Điều 1. Sửa đổi, bổ sung thông tin trong trang 2 và trang 3 mẫu hộ chiếu ngoại giao (mẫu HCNG); mẫu hộ chiếu công vụ (mẫu HCCV); mẫu hộ chiếu phổ thông (mẫu HCPT) ban hành kèm theo Thông tư số 73/2021/TT-BCA ngày 29/6/2021 quy định về mẫu hộ chiếu, giấy thông hành và các biểu mẫu liên quan. +Điều 2. Sửa đổi mẫu Tờ khai đề nghị cấp hộ chiếu phổ thông ở trong nước (mẫu TK01); Tờ khai đề nghị cấp hộ chiếu phổ thông ở nước ngoài (mẫu TK02); Tờ khai đề nghị xác nhận nhân thân cho công dân Việt Nam ở nước ngoài bị mất hộ chiếu (mẫu TK03); Tờ khai đề nghị khôi phục giá trị sử dụng hộ chiếu phổ thông (mẫu TK04); Đơn trình báo mất hộ chiếu phổ thông (mẫu TK05); Văn bản thông báo về việc tiếp nhận đơn báo mất hộ chiếu phổ thông (mẫu VB01); Văn bản của cơ quan đại diện Việt Nam ở nước ngoài đề nghị xác minh nhân thân để cấp hộ chiếu phổ thông cho công dân Việt Nam (mẫu VB02) ban hành kèm theo Thông tư số 73/2021/TT-BCA ngày 29/6/2021 quy định về mẫu hộ chiếu, giấy thông hành và các biểu mẫu liên quan. +Điều 3. Điều khoản thi hành và quy định chuyển tiếp +1. Thông tư này có hiệu lực kể từ ngày 01 tháng 01 năm 2023. +2. Hộ chiếu được cấp trước ngày 01 tháng 01 năm 2023 theo các mẫu đã ban hành trước đây vẫn có giá trị sử dụng đến hết thời hạn ghi trong hộ chiếu./. + +Nơi nhận: +- Thủ tướng Chính phủ (để báo cáo); +- Bộ Ngoại giao, Bộ Quốc phòng (phối hợp thực hiện); +- Các Bộ, cơ quan ngang Bộ, cơ quan thuộc Chính phủ; +- Ủy ban Quốc phòng và An ninh của Quốc hội; +- Ủy ban nhân dân các tỉnh, thành phố trực thuộc TƯ; +- Các đồng chí Thứ trưởng Bộ Công an; +- Các đơn vị trực thuộc Bộ Công an; +- Công an tỉnh, thành phố trực thuộc Trung ương; +- Công báo, Cổng thông tin điện tử Chính phủ, +- Cổng thông tin điện tử Bộ Công an; +- Lưu: VT, QLXNC(P5). BỘ TRƯỞNG + + + + +Đại tướng Tô Lâm + + Mẫu VB02 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12/2022 của Bộ Công an + +……(1)…… +------- CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- +Số: …../………… +V/v xác minh để cấp hộ chiếu phổ thông cho công dân Việt Nam ……., ngày…. tháng….. năm …….. + +Ảnh +(2) CÔNG ĐIỆN +Kính gửi: - Cục Lãnh sự, Bộ Ngoại giao; +- Cục Quản lý xuất nhập cảnh, Bộ Công an. + + +…………(1)……………… trân trọng đề nghị quý Cục cho biết ý kiến về việc cấp hộ chiếu phổ thông của người có thông tin sau: +Họ………………………………. Chữ đệm và tên………………………………………………….. (3) +Giới tính: Nam □ Nữ □ +Ngày sinh……………………… Nơi sinh ……………………………………………………………. +Địa chỉ cư trú ở nước ngoài ………………………………………………………………………….. +Địa chỉ thường trú tại Việt Nam trước khi xuất cảnh ………………………………………………. +……………………………………………………………………………………………………………. +Số điện thoại liên hệ với thân nhân ở trong nước (nếu có) ……………………………………….. +Rời Việt Nam ngày……. /………. / ……………………………………………………………………. +Họ và tên bố……………………………………………… ngày sinh…. /….. /………………….. +Họ và tên mẹ………………………………………………… ngày sinh…. /….. /………………….. +Họ và tên vợ/chồng…………………………………….. ngày sinh….. /…… /........................ +Giấy tờ liên quan do Việt Nam cấp (nếu có)(4):………………………………………………… +…………………………………………………………………………………………………………. +Lý do đề nghị cấp hộ chiếu ……………………………………………………………………… +Xin trao đổi quý Cục để phối hợp công tác./. + +Nơi nhận: +- Như trên; +- Lưu... NGƯỜI CÓ THẨM QUYỀN +(ký, ghi rõ họ và tên, chức vụ, đóng dấu) +Ghi chú: +(1) Tên Cơ quan đại diện Việt Nam tại nước ngoài. +(2) Ảnh mới chụp không quá 06 tháng, cỡ 4cm x 6cm, mặt nhìn thẳng, đầu để trần, rõ mặt, rõ hai tai, không đeo kính, trang phục lịch sự, phông ảnh nền trắng. +(3) Cơ quan đại diện Việt Nam tại nước ngoài ghi rõ họ, chữ đệm và tên của người đề nghị cấp hộ chiếu. +(4) Ghi tên giấy tờ, ngày cấp, cơ quan cấp và gửi kèm. + + Mẫu VB01 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12/2022 của Bộ Công an + +…… (1)……. +------- CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- +Số: ………/………….. …………., ngày……… tháng…. năm ……… + + +THÔNG BÁO +Về việc chuyển đơn trình báo mất hộ chiếu phổ thông +Kính gửi: Cục Quản lý xuất nhập cảnh, Bộ Công an + +Ngày…… /…… /……… ,…………. (1)…………….. tiếp nhận đơn trình báo mất hộ chiếu của người có nhân thân như sau: +Họ……………………………………… Chữ đệm và tên:………………………………………… (2) +Giới tính: Nam □ Nữ □ +Sinh ngày:….. /….. /…….. Nơi sinh (tỉnh, Tp): ……………………………………………………. +Số ĐDCN/CMND: ……………………………………………………………………………………… +……… (1)………………………. chuyển đơn để quý Cục xử lý theo quy định (kèm theo)./. + +Nơi nhận: +- Như trên; +- Người gửi đơn; +- Lưu: …….. NGƯỜI CÓ THẨM QUYỀN +(ký, ghi rõ họ và tên, chức vụ, đóng dấu) +Ghi chú: +(1) Cơ quan tiếp nhận đơn trình báo mất hộ chiếu. +(2) Cơ quan tiếp nhận ghi rõ họ, chữ đệm và tên của người có đơn trình báo mất hộ chiếu. + + Mẫu TK05 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12/2022 của Bộ Công an + +CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- +ĐƠN TRÌNH BÁO MẤT HỘ CHIẾU PHỔ THÔNG (1) +Kính gửi:……………………………(2)………………………. + +1. Họ…………………… Chữ đệm và tên…………………… (3) 2. Giới tính: Nam □ Nữ □ +3. Sinh ngày……. tháng……. năm……………… Nơi sinh (tỉnh, tp) ……………………………… +4. Số ĐDCN/CMND (nếu có) Ngày cấp:…../……./…………. +3. Nơi cư trú hiện nay ………………………………………………………………………………….. +…………………………………………………………………………………………………………… +5. Số điện thoại: …………………………………………………………………………………….. +6. Thông tin về hộ chiếu bị mất(4): +Số hộ chiếu:……………………………. ngày cấp……………. /……………… / ……………….. +Cơ quan cấp: ………………………………………………………………………………………… +8. Hộ chiếu trên đã bị mất vào hồi:……. giờ…….. phút, ngày……….. /……… /……… +9. Hoàn cảnh và lý do cụ thể bị mất hộ chiếu: +………………………………………………………………………………………………………… +………………………………………………………………………………………………………… +……………………………………………………………………………………………………… +………………………………………………………………………………………………………… +……………………………………………………………………………………………………… +Tôi xin cam đoan những thông tin trên là đúng sự thật. + +Xác nhận của Trưởng Công an phường, xã, thị trấn(5) +(Ký và ghi rõ họ và tên, chức vụ, đóng dấu) Làm tại …………ngày......tháng…. năm………. +Người trình báo +(Ký và ghi rõ họ và tên) +Ghi chú: +(1) Người đề nghị điền đầy đủ các nội dung quy định trong mẫu. +(2) Cơ quan Quản lý xuất nhập cảnh nơi thuận lợi hoặc cơ quan Công an nơi gần nhất hoặc đơn vị kiểm soát xuất nhập cảnh tại cửa khẩu hoặc cơ quan đại diện Việt Nam ở nước ngoài nơi thuận lợi. +(3) Họ, chữ đệm và tên viết bằng chữ in hoa. +(4) Trường hợp nhớ chính xác thông tin về hộ chiếu bị mất thì ghi, nếu không nhớ chính xác thì không ghi. +(5) Trưởng Công an phường, xã, thị trấn nơi công dân đang cư trú hoặc nơi báo mất hộ chiếu xác nhận thông tin nhân thân của người viết đơn nếu người báo mất có nhu cầu gửi đơn đến cơ quan Quản lý xuất nhập cảnh qua đường bưu điện. + + Mẫu TK04 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12/2022 của Bộ Công an +CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- +TỜ KHAI ĐỀ NGHỊ KHÔI PHỤC HỘ CHIẾU +(Dùng cho công dân Việt Nam đề nghị khôi phục giá trị sử dụng của hộ chiếu phổ thông bị mất ở trong nước)(1) + +1. Họ………………….. Chữ đệm và tên………………….. (2) 2. Giới tính: Nam □ Nữ □ +3. Sinh ngày……. tháng….. năm…………….. Nơi sinh (tỉnh, Tp) ……………………………… +4. Số ĐDCN/CMND (nếu có) Ngày cấp:…../……./…………. +5. Nơi cư trú hiện tại …………………………………………………………………………………… +6. Số điện thoại: ………………………………………………………………………………………… +7. Thông tin về hộ chiếu đề nghị khôi phục: +Số hộ chiếu:……………………………………… ngày cấp…………… /…………… /…………… +Thời hạn:…………… /…………… /…………… Cơ quan cấp: …………………………………… +8. Lý do đề nghị khôi phục hộ chiếu(3): +…………………………………………………………………………………………………………… +…………………………………………………………………………………………………………… +…………………………………………………………………………………………………………… +………………………………………………………………………………………………………… +…………………………………………………………………………………………………………… +Tôi xin cam đoan những thông tin trên là đúng sự thật./. + + Làm tại ……, ngày …..tháng….. năm………….. +Người đề nghị +(ký, ghi rõ họ tên) +Ghi chú: +(1) Người đề nghị điền đầy đủ thông tin ghi trong mẫu. +(2) Họ, chữ đệm và tên viết bằng chữ in hoa. +(3) Ghi rõ lý do, thời gian, địa điểm, hoàn cảnh... bị mất, tìm lại được hộ chiếu. + + Mẫu TK03 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12/2022 của Bộ Công an + + CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- Ảnh +(2) + TỜ KHAI +(Đề nghị xác nhận nhân thân cho công dân Việt Nam ở nước ngoài bị mất hộ chiếu)(1) + +A. Thông tin người đề nghị: +1. Họ………………………….. Chữ đệm và tên ……………………..(3) 2. Giới tính: Nam □ Nữ □ +3. Sinh ngày…… tháng……….. năm………………. 4. Nơi sinh (tỉnh, TP)…………………… +5. Số định danh cá nhân hoặc CMND Ngày cấp:…../……./…………. +6. Địa chỉ cư trú ……………………………………………………………………………………….. +7. Số điện thoại ………………………………………………………………………………………… +B. Thông tin về thân nhân ở nước ngoài bị mất hộ chiếu +1. Họ…………………………….. Chữ đệm và tên …………………… 2. Giới tính: Nam □ Nữ □ +3. Sinh ngày……… tháng……. năm………… 4. Nơi sinh (tỉnh, TP)………………………… +5. Số ĐDCN/CMND (nếu có) Ngày cấp:…../……./…………. +6. Địa chỉ thường trú ở trong nước trước khi xuất cảnh: ……………………………………… +………………………………………………………………………………………………………… +7. Địa chỉ ở nước ngoài…………………………………………………………………………... +……………………………………………………………………………………………………… +8. Xuất cảnh Việt Nam ngày …../…… /….. qua cửa khẩu ……………………………………… bằng hộ chiếu số…………………………………. cấp ngày….. /….. / …………………………. +9. Dự kiến về Việt Nam ngày …../…… / ……………….. +10. Giấy tờ chứng minh quan hệ với thân nhân ở nước ngoài bị mất hộ chiếu(5): ……………… +11. Nội dung đề nghị: Cục Quản lý xuất nhập cảnh, Bộ Công an xác nhận ảnh và thông tin nhân thân để thân nhân tôi được cấp hộ chiếu phổ thông tại …………………………….(6) +Tôi cam đoan những nội dung khai trên đây là đúng và chịu trách nhiệm trước pháp luật./. + + Làm tại ………..ngày….. tháng..... năm ……. +Người đề nghị +(Ký, ghi rõ họ và tên) +Ghi chú: +(1) Người đề nghị điền đầy đủ thông tin ghi trong mẫu. +(2) Ảnh mới chụp của công dân Việt Nam ở nước ngoài bị mất hộ chiếu, cỡ 4cm x 6cm, mặt nhìn thẳng, đầu để trần, rõ mặt, rõ hai tai, không đeo kính, trang phục lịch sự, phông ảnh nền trắng. Dán 01 ảnh vào khung phía trên, kèm theo 01 ảnh để rời. +(3) (4) Họ, chữ đệm và tên viết bằng chữ in hoa. +(5) Trường hợp không có giấy tờ chứng minh phải có bản giải trình. +(6) Ghi tên cơ quan đại diện Việt Nam ở nước ngoài nơi cấp hộ chiếu. + + Mẫu TK02 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12/2022 của Bộ Công an + +CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- Ảnh +(2) +TỜ KHAI +(Dùng cho công dân Việt Nam đề nghị cấp hộ chiếu phổ thông ở nước ngoài)(1) + +1. Họ………………………. Chữ đệm và tên……………………….. (3) 2. Giới tính: Nam □ Nữ □ +3. Sinh ngày………. tháng……. năm………… Nơi sinh(4) (tỉnh, TP)…………………………. +4. Số ĐDCN/CMND (nếu có) Ngày cấp:…../……./…………. +5. Dân tộc………….. 6. Tôn giáo……………….. 7. Số điện thoại(5)............................................. +8. Địa chỉ cư trú ở nước ngoài …………………………………………………………………… +………………………………………………………………………………………………………… +9. Địa chỉ thường trú ở trong nước trước khi xuất cảnh……………………………………….. +…………………………………………………………………………………………………………. +10. Nghề nghiệp………………………… 11. Tên và địa chỉ cơ quan (nếu có).………………… +12. Cha: họ và tên……………………………………… sinh ngày…../……… /…………………. +Mẹ: họ và tên……………………………………… sinh ngày….. /……… /……………….. +Vợ /chồng: họ và tên……………………………………… sinh ngày …../...../…………….. +13. Hộ chiếu phổ thông lần gần nhất (nếu có) số……………………… cấp ngày….. /….. /...... +14. Nội dung đề nghị(6) ……………………………………………………………………………… +Cấp hộ chiếu có gắn chip điện tử □ Cấp hộ chiếu không gắn chip điện tử □ +Tôi xin cam đoan những thông tin trên là đúng sự thật. + + Làm tại…………… ngày….. tháng..... năm……. +Người đề nghị(7) +(Ký, ghi rõ họ và tên) + +Ảnh +(2) Chú thích: +(1) Người đề nghị điền đầy đủ thông tin ghi trong mẫu, không được thêm bớt. +(2) Ảnh mới chụp không quá 06 tháng, cỡ 4cm x 6cm, mặt nhìn thẳng, đầu để trần, rõ mặt, rõ hai tai, không đeo kính, trang phục lịch sự, phông ảnh nền trắng. +(3) Họ, chữ đệm và tên viết bằng chữ in hoa. +(4) Nếu sinh ra ở nước ngoài thì ghi tên quốc gia. +(5) Ghi số điện thoại liên lạc ở nước ngoài và số điện thoại của thân nhân thường xuyên liên hệ ở Việt Nam (nếu có). +(6) Ghi cụ thể: Đề nghị cấp hộ chiếu lần đầu hoặc từ lần thứ hai; đề nghị khác nếu có (ghi rõ lý do). Trường hợp đề nghị cấp hộ chiếu có (hoặc không) gắn chip điện tử thì đánh dấu (X) vào ô tương ứng. +(7) Đối với người mất năng lực hành vi dân sự, có khó khăn trong nhận thức và làm chủ hành vi, người chưa đủ 14 tuổi thì người đại diện hợp pháp ký thay. + + Mẫu TK01 +Ban hành kèm theo Thông tư số 68/2022/TT-BCA ngày 31/12 /2022 của Bộ Công an + + CỘNG HÒA XÃ HỘI CHỦ NGHĨA VIỆT NAM +Độc lập - Tự do - Hạnh phúc +--------------- Ảnh +(2) + TỜ KHAI +(Dùng cho công dân Việt Nam đề nghị cấp hộ chiếu phổ thông ở trong nước)1) + +1. Họ…………….. Chữ đệm và tên…………………..(3) 2. Giới tính: Nam □ Nữ □ +3. Sinh ngày………. tháng…………. năm………… Nơi sinh(4) (tỉnh, TP)……………………… +4. Số ĐDCN/CMND (nếu có) Ngày cấp:…../……./…………. +5. Dân tộc………………………… 6. Tôn giáo ………….7. Số điện thoại……………………… +8. Địa chỉ đăng ký thường trú …………………………………………………………………… +…………………………………………………………………………………………………….. +………………………………………………………………………………………………………… +9. Địa chỉ đăng ký tạm trú……………………………………………………………………………… +………………………………………………………………………………………………………… +10. Nghề nghiệp………………………11. Tên và địa chỉ cơ quan (nếu có)…………………… +12. Cha: họ và tên ……………………………………………… sinh ngày …./….. / ………………. +Mẹ: họ và tên ……………………………………………… sinh ngày….. /….. / ……………… +Vợ /chồng: họ và tên ……………………………………………… sinh ngày …./…../……….. +13. Hộ chiếu PT lần gần nhất (nếu cố) số ………………………………cấp ngày …../…../………. +14. Nội dung đề nghị(5) ……………………………………………………………………………… +Cấp hộ chiếu không có gắn chip điện tử □ Cấp hộ chiếu có gắn chip điện tử □ +Tôi xin cam đoan những thông tin trên là đúng sự thật. + +Xác nhận của Trưởng Công an phường/xã/thị trấn(6) +(Ký, ghi rõ họ và tên, chức vụ, đóng dấu) ………… , ngày….. tháng..... năm…… +Người đề nghị(7) +(Ký, ghi rõ họ và tên) + +Ảnh +(2) Chú thích: +(1) Người đề nghị điền đầy đủ thông tin ghi trong mẫu, không được thêm bớt. +(2) Ảnh mới chụp không quá 06 tháng, cỡ 4cm x 6cm, mặt nhìn thẳng, đầu để trần, rõ mặt, rõ hai tai, không đeo kính, trang phục lịch sự, phông ảnh nền trắng. +(3) Họ, chữ đệm và tên viết bằng chữ in hoa. +(4) Nếu sinh ra ở nước ngoài thì ghi tên quốc gia. +(5) Ghi cụ thể: Đề nghị cấp hộ chiếu lần đầu hoặc từ lần thứ hai; đề nghị khác nếu có (ghi rõ lý do). Trường hợp đề nghị cấp hộ chiếu có (hoặc không) gắn chip điện tử thì đánh dấu (X) vào ô tương ứng. +(6) Áp dụng đối với người mất năng lực hành vi dân sự, người có khó khăn trong nhận thức và làm chủ hành vi, người chưa đủ 14 tuổi. Trưởng Công an phường, xã, thị trấn nơi thường trú hoặc tạm trú xác nhận về thông tin điền trong tờ khai và ảnh dán trong tờ khai là của một người; đóng dấu giáp lai vào ảnh dán ở khung phía trên của tờ khai. +(7) Đối với người mất năng lực hành vi dân sự, có khó khăn trong nhận thức và làm chủ hành vi, người chưa đủ 14 tuổi thì người đại diện hợp pháp ký thay. \ No newline at end of file diff --git a/graphrag-ollama-config/input/text_bibin_1571bcfc.txt b/graphrag-ollama-config/input/text_bibin_1571bcfc.txt new file mode 100644 index 0000000..5e1bf9b --- /dev/null +++ b/graphrag-ollama-config/input/text_bibin_1571bcfc.txt @@ -0,0 +1,414 @@ +# Source: DIRECT TEXT INPUT +# Title: bibin +# Ingested: 2026-01-13 14:04:16 +# Word Count: 3376 +# Character Count: 25426 + +--- +EB1A (Green Card) criteria for "Aliens of Extraordinary Ability." Below is an +analysis of +how you meet the 10 USCIS criteria (you need to satisfy at least 3): +1. Receipt of Lesser Nationally or Internationally Recognized Prizes or Awards +Microsoft MVP- AI Platforms +UAE Golden Visa (Talented Person Category) – A prestigious award granted to +individuals with exceptional contributions in their field. +2. Membership in Associations Requiring Outstanding Achievements +IEEE Senior Membership – Requires at least 10 years of experience and significant +professional contributions. +Raptors Fellowship +CDMP (Certified Data Management Professional) – A globally recognized +certification by DAMA International. +IET Chartered Engineer (CEng) +BCS Fellowship +3. Published Material About You in Professional or Major Media +IBTimes.sg - Bibin Prathap Launches Space AI: Pioneering a New Era of Trustworthy +Enterprise Intelligence +https://www.ibtimes.sg/bibin-prathap-launches-space-ai-pioneering-new-era-trustworth +y-enterprise-intelligence-82023 +https://www.meetup.com/chicago-data-and-ai/ - +https://www.meetup.com/chicago-data-and-ai/events/311085114/?eventOrigin=group_past_ +events&_gl=1*darjxc*_up*MQ..*_ga*MTMzMjA2MTE0Ni4xNzYzNTU1OTk1*_ga_NP82XMKW0P*czE3NjM +1NTU5OTUkbzEkZzAkdDE3NjM1NTU5OTUkajYwJGwwJGgw +https://www.devopschat.co/meetings/building-enterprise-ai-that-works - +https://globeeawards.com/2025-judges-impact-awards/ +https://committers.top/uae.html +https://www.falconebiz.com/director/10951196/BIBIN-PRATHAP +https://www.aict.info/2025/info/AICT2025-Conference-Program.pdf - presenter +https://cdn.adu.ac.ae/images-container/docs/default-source/icasf-docs/icasf-2025---a +genda.pdf - presenter +https://youtu.be/pRkVJW4sGfo?si=X6cdYUZOG8tt-SDU +4. Participation as a Judge of the Work of Others +IEEE Senior Membership Review Panel – Serving as a reviewer for senior +membership applications demonstrates judging authority. +Pull request review of Famous pubic repositories +https://github.com/bibinprathap/whatsapp-chatbot/pull/7 +https://github.com/bibinprathap/VeritasGraph/pulls +5. Original Scientific, Scholarly, or Business-Related Contributions +WhatsApp Chatbot (Open-Source, 97 GitHub Stars, 57 Forks) – Recognized +technical contribution in conversational AI. +https://github.com/bibinprathap/whatsapp-chatbot +https://github.com/bibinprathap/VeritasGraph +https://github.com/bibinprathap/erp +6. Authorship of Scholarly Articles +https://ieeexplore.ieee.org/xpl/conhome/1003017/all-proceedings - +A Data-Driven Framework for Urban Mobility: +Real-Time Traffic Analysis in Dubai Using Google +Maps API and R +https://www.adu.ac.ae/conferences-competitions/international-conference-on-advancing +-sustainable-futures/the-3rd-international-conference-on-advancing-sustainable-futur +es +VeritasGraph: A Sovereign GraphRAG Framework for Enterprise-Grade AI with Verifiable +Attribution +https://bibinprathap.medium.com/ +Medium Articles (2,500+ views, 530+ reads on top article) – Demonstrates +dissemination of technical knowledge. +7. Display of Work at Artistic Exhibitions or Showcases +8. Performance in a Leading or Critical Role for Distinguished Organizations +Senior Specialist, Data management, Abu Dhabi Executive Office – Led AI initiatives +for government +operations. +Founder & CEO, SpaceAI App Private Limited – Leadership in AI innovation. +Founder & CTO, – https://pmspace.ai/ Leadership in AI innovation. +Founder & CTO, – https://space-sign.ai/ , california ,USA, Leadership in AI +innovation. +Senior AI Consultant, Tech Mahindra – Directed AI projects for Abu Dhabi +9. High Salary or Remuneration Relative to Peers +8441 USD - as a senior specialist , data Management for 4 years +10. Commercial Success in the Performing Artistic +VIA PREMIUM PROCESSING +USCIS +Attn: I-140 +RE: I-140 Immigrant Petition for Alien Worker +Petitioner/Beneficiary: Bibin PRATHAP +Classification: EB-1A Alien of Extraordinary Ability (INA § 203(b)(1)(A)) +Field of Endeavor: Enterprise AI Architecture and Verifiable Attribution Systems +TO THE HONORABLE ADJUDICATOR: +This legal memorandum and the accompanying exhibits are submitted in support of the +I‑140 Immigrant Petition for Alien Worker on behalf of Mr. Bibin Prathap +(“Petitioner”), who seeks classification as an alien of extraordinary ability in the +field of Enterprise AI Architecture and Verifiable Attribution Systems. +Mr. Prathap is among the small percentage of professionals who have risen to the +very top of this field. See 8 C.F.R. § 204.5(h)(2). His career is marked by a +sustained record of pioneering “systems of intelligence” that directly address one +of the most pressing challenges in contemporary Artificial Intelligence: ensuring +trust, auditability, security, and verifiable attribution in Large Language Model +(LLM)–based enterprise systems. +Under the framework articulated in Kazarian v. USCIS, 596 F.3d 1115 (9th Cir. 2010), +this petition demonstrates, by a preponderance of the evidence, that Mr. Prathap +satisfies multiple regulatory criteria at 8 C.F.R. § 204.5(h)(3) and, under a final +merits analysis, has achieved sustained national and international acclaim. His +recognition includes, among other things, the Microsoft Most Valuable Professional +(MVP) Award in AI Platforms, elevation to Senior Member of the IEEE, and leadership +in architecting AI‑driven digital infrastructure for sovereign government entities. +He now serves as Founder and CTO of California‑based Space‑Sign.ai, where he +advances U.S. technological interests by developing AI architectures that keep +enterprise data private, verifiable, and secure while providing attribution‑aware, +audit‑ready LLM outputs. +I. PROPOSED ENDEAVOR IN THE UNITED STATES +Mr. Prathap intends to continue his work in the United States as Founder & CTO of +Space‑Sign.ai, headquartered in California. His proposed endeavor centers on the +design and deployment of VeritasGraph, a graph‑native, retrieval‑augmented framework +that enforces “verifiable attribution” for enterprise AI systems. +Alignment with U.S. National Priorities: +The United States is actively working to mitigate the “black box” nature of AI +systems and to strengthen accountability, safety, and provenance in automated +decision‑making. Executive Order 14110 directs the development of “effective +labeling and content provenance mechanisms” and emphasizes secure, trustworthy AI +systems that enable auditing and traceability. Mr. Prathap’s work directly advances +these goals. By integrating knowledge graphs with Retrieval‑Augmented Generation +(RAG), he builds AI systems that can identify, trace, and document their underlying +sources—capabilities that are especially critical in regulated environments such as +finance, healthcare, public administration, and defense‑adjacent use cases. +Through Space‑Sign.ai and related initiatives (including VeritasGraph and +sovereign‑grade enterprise RAG architectures), Mr. Prathap’s continued work in the +United States will materially enhance U.S. AI sovereignty, data security, and +compliance‑grade trust in AI‑mediated workflows. +II. TIER 1 ANALYSIS: SATISFACTION OF REGULATORY CRITERIA +The Petitioner submits evidence that he satisfies more than the required three +criteria set forth in 8 C.F.R. § 204.5(h)(3), including awards, membership, judging, +original contributions, leading/critical roles, and high remuneration. +A. Criterion 1: Lesser Nationally or Internationally Recognized Prizes or Awards +Regulation: 8 C.F.R. § 204.5(h)(3)(i) – Documentation of the alien’s receipt of +lesser nationally or internationally recognized prizes or awards for excellence in +the field of endeavor. +1. Microsoft Most Valuable Professional (MVP) Award – AI Platforms +Mr. Prathap has been selected as a Microsoft Most Valuable Professional (MVP) in the +category of AI Platforms. +Significance: The Microsoft MVP Award is a globally recognized distinction conferred +by Microsoft on external community leaders and technical experts. It is not a +certification, employment benefit, or paid program; it is an independent recognition +of sustained technical excellence and substantial community impact. +Selectivity: Microsoft reports that, worldwide, only a small population of +technology professionals hold active MVP status across all categories +combined—estimated at fewer than several thousand individuals relative to millions +of developers and IT professionals in the Microsoft ecosystem. Within the AI +Platforms specialty, the cohort is even more limited, evidencing an elite level of +recognition. +Selection Criteria: Candidates are nominated and then vetted by Microsoft product +teams and community program managers. Evaluation focuses on demonstrated technical +depth, real‑world impact, and documented community contributions (e.g., talks, +open‑source work, publications) over the preceding 12 months. This process confirms +that recipients are among the leading experts in their technology domain. +2. UAE Golden Visa – Talented Person Category +Mr. Prathap has been granted the United Arab Emirates (UAE) Golden Visa in the +“Talented Person” category. +Significance: The UAE Golden Visa in this classification is awarded to individuals +with exceptional specialized talent and recognized contributions to their respective +fields. It is granted following review and endorsement by relevant governmental +authorities and reflects governmental recognition of the recipient as a professional +of national importance, particularly in innovation‑driven sectors such as advanced +technology and AI. +Use in Petition: While the primary focus of this criterion is on the Microsoft MVP +Award, the UAE Golden Visa serves as corroborating evidence that a sovereign +government has independently recognized Mr. Prathap’s exceptional standing and +contribution in technology and AI. +B. Criterion 2: Membership in Associations Requiring Outstanding Achievements +Regulation: 8 C.F.R. § 204.5(h)(3)(ii) – Documentation of the alien’s membership in +associations in the field for which classification is sought, which require +outstanding achievements of their members, as judged by recognized national or +international experts. +1. Senior Member of the IEEE (Institute of Electrical and Electronics Engineers) +Mr. Prathap has been elevated to the grade of Senior Member of IEEE, the world’s +largest professional association for the advancement of technology. +Outstanding Achievement Requirement: +Standard IEEE membership is open to individuals who meet basic educational or +professional prerequisites. In contrast, Senior Member grade is reserved for members +who satisfy all of the following: +- At least 10 years of professional practice in IEEE‑designated fields, and +- At least 5 years of demonstrated “significant performance” documented through +achievements such as technical leadership, innovation, and impact on the profession. +Judgment by Experts: +Senior Member elevation requires: +- A formal application detailing the candidate’s experience and achievements; +- Endorsements from existing Senior Members or Fellows; and +- Approval by the IEEE Admission and Advancement Committee, a panel of recognized +experts who evaluate eligibility under objective standards. +Exclusivity: +Publicly available IEEE statistics show that only a minority of IEEE’s 400,000+ +members hold the Senior Member or Fellow grades. Senior Member status is therefore +limited and indicates a recognized record of professional excellence and +contribution to the field at an international scale. +2. Fellow of the British Computer Society (FBCS) and IET Chartered Engineer (CEng) +Mr. Prathap has been elected a Fellow of BCS, The Chartered Institute for IT, and +has been granted Chartered Engineer (CEng) status through the Institution of +Engineering and Technology (IET). +Fellow of BCS (FBCS): +Fellowship in BCS is the highest professional grade and is restricted to individuals +who can demonstrate: +- A sustained record of leadership, authority, and influence in the IT profession; +- Significant responsibility and contribution to the advancement of computing +practice; and +- Endorsement and validation by senior peers and review committees within the +institution. +IET Chartered Engineer (CEng): +Chartered Engineer is an internationally recognized professional registration that +confirms: +- Advanced engineering competence and deep technical expertise; +- Responsibility for complex systems and critical engineering decisions; and +- Assessment by nationally and internationally recognized experts under standards +aligned with the UK Engineering Council. +Both FBCS and CEng require verifiable records of outstanding professional +achievements and are not available through payment of dues alone. +C. Criterion 4: Participation as a Judge of the Work of Others +Regulation: 8 C.F.R. § 204.5(h)(3)(iv) – Evidence of the alien’s participation, +either individually or on a panel, as a judge of the work of others in the same or +an allied field of specialization. +1. IEEE Senior Membership Review and Related Evaluation Activities +As an experienced Senior Member, Mr. Prathap has participated in the evaluation of +professional materials and applicants in alignment with IEEE review processes, +contributing expert assessment of whether other engineers and technologists meet the +threshold for significant performance and elevation. +Nature of Judging: +In this capacity, he evaluates the technical and professional qualifications of +other professionals, including their career trajectory, leadership roles, and impact +on the field. This evaluation influences whether these individuals are recognized at +advanced grades within IEEE or allied bodies. Only recognized experts are entrusted +with such evaluative responsibilities. +2. Maintainer and Reviewer of Open‑Source AI and Data Systems Repositories +Mr. Prathap is the principal maintainer and reviewer of several public, open‑source +repositories, including but not limited to: +- “whatsapp‑chatbot” (open‑source conversational AI framework); and +- “VeritasGraph” (graph‑based RAG and verifiable attribution framework). +Technical Adjudication: +As maintainer, he: +- Reviews and adjudicates pull requests from external contributors; +- Evaluates the technical correctness, security posture, scalability, and +architectural soundness of submitted code; and +- Exercises final authority to approve (merge) or reject changes. +These repositories serve hundreds of users, and their evolution is directly shaped +by his judgment of third‑party technical work. This activity constitutes ongoing, +public, expert review of the work of others in an allied field (enterprise AI, data +systems, and software engineering). +D. Criterion 5: Original Contributions of Major Significance +Regulation: 8 C.F.R. § 204.5(h)(3)(v) – Evidence of the alien’s original scientific, +scholarly, artistic, athletic, or business‑related contributions of major +significance in the field. +1. VeritasGraph: A Sovereign GraphRAG Framework with Verifiable Attribution +Mr. Prathap is the creator and principal architect of VeritasGraph, a framework that +integrates knowledge graphs with Retrieval‑Augmented Generation (RAG) to deliver +“verifiable attribution” in enterprise AI. +Originality: +Traditional LLM deployments often function as opaque systems, with limited ability +to trace outputs back to underlying sources. VeritasGraph introduces: +- A graph‑based knowledge representation layer that encodes entities, relationships, +and provenance; +- A retrieval pipeline that constrains LLM responses to curated, attributed +knowledge; and +- An attribution mechanism that surfaces source nodes and relationships used in +generating each answer. +This approach goes beyond standard RAG by explicitly structuring and exposing +provenance, thereby addressing the AI “hallucination” and auditability problem in a +novel, architectural manner. +Major Significance: +Evidence of impact and significance includes: +- Adoption and use in enterprise contexts where data sovereignty, compliance, and +audit trails are mandatory; +- Public interest reflected in repository metrics (stars, forks, and external +references), indicating that independent developers and organizations are building +upon his work; and +- Integration of VeritasGraph principles in solutions delivered to Abu Dhabi +government entities, enabling secure, sovereign deployment of AI systems that must +withstand regulatory and security scrutiny. +2. Mobile Inspection Management System and Related Government AI Systems +While serving in senior roles on Abu Dhabi government initiatives, Mr. Prathap led +the design and implementation of the Mobile Inspection Management System for the Abu +Dhabi Municipality and related AI‑enabled data management systems. +Significance: +These systems transformed previously manual or siloed processes into integrated, +intelligent platforms by: +- Digitizing critical municipal workflows and inspections; +- Integrating national identity verification (e.g., Emirates ID) with computer +vision (e.g., Google Vision API) and analytics; and +- Enabling real‑time data collection, oversight, and decision‑support for a major +world capital. +These contributions had material impact on governmental efficiency, data quality, +and service delivery, and they embody “systems of intelligence” that other +jurisdictions and organizations look to as benchmarks for digital transformation. +E. Criterion 8: Leading or Critical Role for Distinguished Organizations +Regulation: 8 C.F.R. § 204.5(h)(3)(viii) – Evidence that the alien has performed in +a leading or critical role for organizations or establishments that have a +distinguished reputation. +1. Senior Specialist, Data Management – Abu Dhabi Executive Office +Mr. Prathap served as Senior Specialist, Data Management, at the Abu Dhabi Executive +Office, a sovereign entity that supports the executive leadership of the Emirate of +Abu Dhabi. +Critical Role: +In this position, he: +- Led enterprise data management and AI‑driven analytics initiatives; +- Defined and implemented data strategy and governance underpinning high‑level +government decision‑making; and +- Provided technical leadership for programs that directly supported the Executive +Office’s mandate for digital transformation and evidence‑based policy. +Distinguished Reputation: +The Abu Dhabi Executive Office, as a central governmental body of a major global +capital and one of the world’s most prominent sovereigns, has an internationally +distinguished reputation. Holding a senior technical role in such an entity +underscores the level of trust placed in his expertise. +2. Senior AI Consultant – Tech Mahindra +Mr. Prathap served as Senior AI Consultant for Tech Mahindra, a global IT and +consulting firm with multi‑billion‑dollar annual revenues and a substantial +international client base. +Critical Role: +He was the lead architect and technical driver for high‑visibility AI and data +projects, including the Abu Dhabi Municipality’s digital inspection and intelligence +platforms. In this capacity, he: +- Defined the overall AI architecture and data integration strategy; +- Led multi‑disciplinary teams to deliver production‑grade solutions; and +- Served as the primary technical interface with senior government stakeholders. +His role was central to the success of these flagship engagements, which were +critical to Tech Mahindra’s regional reputation and client relationships. +3. Founder & CEO / CTO – SpaceAI App Private Limited, pmspace.ai, and Space‑Sign.ai +Mr. Prathap has also served as Founder & CEO or Founder & CTO of multiple AI‑focused +companies, including SpaceAI App Private Limited, pmspace.ai, and Space‑Sign.ai +(California, USA). +Leading Role and Organizational Distinction: +In these roles, he: +- Sets the technical vision and product roadmap for enterprise AI solutions; +- Leads R&D and productization of advanced AI architectures, including verifiable +attribution and sovereign RAG systems; and +- Represents the companies in international conferences, meetups, and technical +communities, building their reputation as innovators in trustworthy enterprise AI. +These organizations, while entrepreneurial, operate in an advanced technological +niche and further demonstrate that Mr. Prathap is consistently placed in positions +of top‑level responsibility where his expertise is mission‑critical. +F. Criterion 9: High Salary or Remuneration +Regulation: 8 C.F.R. § 204.5(h)(3)(ix) – Evidence that the alien has commanded a +high salary or other significantly high remuneration for services, in relation to +others in the field. +1. Senior Specialist Compensation – Abu Dhabi Executive Office +In his role as Senior Specialist, Data Management, at the Abu Dhabi Executive +Office, Mr. Prathap received compensation of approximately USD $8,441 per month +(tax‑free), for multiple years. +Comparative Market Analysis: +Independent salary surveys for Abu Dhabi and the broader UAE market (e.g., Michael +Page, Hays, and other professional compensation reports) indicate that: +- Typical monthly salaries for data specialists and comparable roles fall +substantially below this level; and +- The 90th percentile for senior data professionals in Abu Dhabi is generally in the +AED 30,000–35,000 range. +Mr. Prathap’s tax‑free salary places him well above common market benchmarks for +similar positions. When adjusted for tax‑equivalent gross income in major U.S. +markets, his compensation is consistent with or exceeds top‑tier pay levels for +senior data and AI architects. This demonstrates that employers—particularly a +sovereign executive office—have paid a premium for his services, confirming his +standing as a highly sought‑after expert in his field. +III. TIER 2: FINAL MERITS DETERMINATION +Having demonstrated that Mr. Prathap satisfies more than the requisite three +criteria under 8 C.F.R. § 204.5(h)(3), the analysis turns to whether, in the +aggregate, the evidence establishes that he is an individual of extraordinary +ability with sustained national or international acclaim whose achievements have +been recognized in the field. +1. Cross‑Sector Recognition and “Triangulation” of Excellence +The record shows consistent recognition across three independent domains: +- Industry Recognition: Microsoft MVP in AI Platforms recognizes him as an +exceptional technical leader and community contributor in the global Microsoft AI +ecosystem. +- Professional Peer Recognition: Elevation to IEEE Senior Member, BCS Fellowship, +and IET Chartered Engineer confirms that independent professional bodies have +assessed his credentials and judged his achievements to be outstanding. +- Governmental Trust and Distinction: His UAE Golden Visa (Talented Person category) +and senior roles for Abu Dhabi’s Executive Office and Municipality demonstrate +governmental reliance on his expertise in mission‑critical, sovereign contexts. +Few professionals achieve this depth of recognition across major technology vendors, +professional institutions, and sovereign government entities. This triangulation +strongly supports a finding of sustained acclaim. +2. Impact on a Critical Problem: Trustworthy and Verifiable Enterprise AI +The field of AI is at an inflection point: organizations seek productivity gains +from LLMs and generative systems, yet adoption is often constrained by concerns +about hallucinations, security, provenance, and regulatory compliance. Mr. Prathap’s +work directly addresses these barriers through: +- VeritasGraph (graph‑based RAG with verifiable attribution); +- Enterprise‑grade architectures for sovereign, private deployments; and +- Systems of intelligence that convert raw data into auditable, explainable +decisions. +These contributions do not merely implement existing techniques; they define +practical patterns and frameworks that organizations can adopt to safely scale AI. +As such, his contributions are of major significance to the maturation of enterprise +AI as a trustworthy infrastructure layer. +3. Sustained, Multi‑Year Record of Accomplishment +Mr. Prathap’s record is sustained and cumulative, rather than isolated: +- Early work on open‑source projects (e.g., whatsapp‑chatbot) gained international +traction and community use; +- He progressed into senior consulting and architectural roles at a major global IT +services firm (Tech Mahindra), leading high‑impact public sector AI initiatives; +- He then advanced to a senior post at the Abu Dhabi Executive Office, directing +data and AI strategy at the sovereign level; +- In parallel, he earned prestigious professional memberships and awards (IEEE +Senior Member, FBCS, CEng, Microsoft MVP, UAE Golden Visa); and +- He now leads Space‑Sign.ai in California, focusing on sovereign‑grade, +attribution‑aware enterprise AI architectures. +This trajectory evidences a consistent pattern of achievement and recognition over +more than a decade, satisfying the requirement for sustained national and +international acclaim. +IV. CONCLUSION +The totality of the evidence demonstrates that Mr. Bibin Prathap: +- Has risen to the very top of the field of Enterprise AI Architecture and +Verifiable Attribution Systems; +- Has achieved and maintained sustained national and international acclaim; and +- Intends to continue work in the United States that will substantially benefit the +nation by advancing secure, trustworthy, and auditable AI infrastructure. +Accordingly, the Petitioner respectfully requests that USCIS approve the I‑140 +Immigrant Petition for Alien Worker in the EB‑1A classification on behalf of Mr. +Bibin Prathap. +Respectfully submitted, +Attorney for Petitioner \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_5_Free_Online_Courses_From_Harvard_University_To_B_e6a9bd75.txt b/graphrag-ollama-config/input/youtube_5_Free_Online_Courses_From_Harvard_University_To_B_e6a9bd75.txt new file mode 100644 index 0000000..854172d --- /dev/null +++ b/graphrag-ollama-config/input/youtube_5_Free_Online_Courses_From_Harvard_University_To_B_e6a9bd75.txt @@ -0,0 +1,10 @@ +# Source: YOUTUBE +# Title: 5 Free Online Courses From Harvard University To Boost Your Career +# Channel: Shane Hummus +# Video ID: ssmlfH0bmi0 +# Duration: 14 minutes +# Ingested: 2026-01-18 13:09:27 +# Word Count: 3099 + +--- +today we're going to be diving into something that might just blow your mind imagine getting a Harvard Education for free no this isn't a glitch in The Matrix and no you don't need a time machine or a small fortune we're talking about five incredible free online courses from Harvard that could Skyrocket your careers faster than you can say Veritas so whether you're a tech Enthusiast a data Dynamo or a Wordsmith in the making there's something here for you but before we dive in let's address the elephant in the room yes these are courses that are actually from Harvard yes they're completely free and no you don't need to be a genius to take them all you need is curiosity and the willingness to learn so buckle up grab your favorite note ticking tool and let's embark on this ivy league Adventure kicking off our list is the crown jewel of Harvard's online offerings which is the cs50 introduction to computer science now I know what some of you might be thinking computer science isn't that just for Tech Geeks well hold on to your hoodies because this course is about to change your perception faster than you can say hello world cs50 is not just a course it's a phenomenon it's Harvard's largest course on campus and has become one of the most popular M's or mukes which stands for massive open online courses globally now according to class Central as of 2023 over 3.7 million people have enrolled in cs50 online that's more than the population of some countries but what makes cs50 so special well it's like the Swiss army knife of Tech courses you'll learn scratch which is a beginner-friendly programming language you'll learn C which is the grandfather of modern programming languages you'll learn python which is the Swiss army knife of coding you'll learn SQL which is the language that makes databases sing plus you're going to learn HTML CSS and JavaScript which is the Holy Trinity of web development and that's just scratching the surface but don't just take my word for it let's hear from someone who's been through the cs50 gauntlet Sarah said before cs50 I thought algorithms were just for Tech Geniuses now I use algorithmic thinking in my everyday life from optimizing my grocery shopping to managing projects at work it's like I've been given a new pair of eyes to see the world now I know what you're thinking this sounds great but is it really for beginners and the answer is a resounding yes David J Milan the charismatic Professor behind cs50 has a knack for making complex Concepts digestible and he once explained memory allocation using a phone book and a saw Yes you heard that right but cs50 isn't just about learning to code it's about developing a problemsolving mindset that can be applied to any field so whether you're a marketer trying to optimize campaigns a teacher developing a curriculum or an entrepreneur building the next big thing the computational thinking skills you'll gain from cs50 are invaluable and the best part you can take this course at your own pace where whether you're a night owl coding at 2: a.m. or an early bird catching the programming worm cs50 fits into your schedule so are you ready to embark on this computational journey remember in the words of Professor Milan what ultimately matters in this course is not so much where you end up relative to your classmates but where you end up relative to yourself when you begin now let's move on to our next course which is going to turn you into a data wizard faster than you can say are you ready welcome to the world of data science specifically R Basics and no we're not talking about courses for pirates although shouting R might become your new favorite thing this course is part of Harvard's professional certificate in data science program and it's your ticket to becoming a data Dynamo but why are you might ask well in the words of Hadley Wickham Chief scientist at R Studio R is a language for people who want to get things done it's not a language for computer scientists it's a language for scientists who compute in other words R is the Swiss army knife of data science it's versatile powerful and once you get the hang of it it's incredibly fun to use so what exactly will you learn from this course well buckle up because we're about to take a deep dive so first of all there's our syntax the grammar of data science then there's data types understanding the building blocks of information then you've got vectors which is your new best friend in data manipulation then you've got sorting because sometimes order matters then you've got data frames which is the PowerHouse of data organization and then you've got data visualization which is turning numbers into I candy now I know what some of you might be thinking data science sounds boring it's just staring at numbers all day right well let me stop you right there data science is like being a detective in the digital age you're uncovering hidden patterns solving complex puzzles and sometimes pred the future so it's kind of like having a superpower don't believe me let's hear from someone who's been through this course this person says before taking the course I thought data science was all about complex mathematics but R Basics showed me that it's really about telling stories with data now I use R in my marketing to analyze campaign performance and it's revolutionized how we make decisions but why should you care about data science well according to the US Bureau of Labor Statistics the employment of data scientists is projected to grow 36% from 2021 to 2031 much faster than the average for all occupation patients that's not just growth that's an explosion and it's not just tech companies that need data scientists from Healthcare to finance from Sports to entertainment every industry is looking for people who can make sense of their data so by learning R you're not just picking up a new skill you're opening doors to countless opportunities but here's the kicker this course isn't just about learning R it's about developing a data driven mindset you'll learn to ask the right questions approach the problem systematically and communicate your findings effectively and these are skills that will serve you well in any career and the best part you can learn all of this for free at your own pace from the comfort of your favorite chair whether you're a complete beginner or someone looking to add another tool to your data science tool cut this course has something for you so are you ready to speak the language of data remember in the world of data science curiosity is your greatest asset and as the famous statistician John Tuki once said the best thing about being a statistician is that you get to play in everyone's backyard now let's switch gears and dive into a field that's Bridging the Gap between the humanities and the digital world so get ready to become a digital Renaissance person welcome to the fast fting world of digital Humanities now I know what you're thinking digital Humanities isn't that like reading Shakespeare on a Kindle well hold on to your ebooks because it's so much more than that digital Humanities is where technology meets culture where algorithms meet art and where data meets storytelling it's like giving the humanities a superpower and this course offered by Harvard University is your ticket to becoming a cultural codebreaker but what exactly will you learn well let's break it down first there's digital tools for Humanity's research which is your new digital Swiss army knife then there's data visualization in Humanities making centuries of culture visually digestible then there's text analysis and Mining which is uncovering hidden patterns and vast amounts of text then you've got digital mapping and spatial analysis which is putting history and culture on the map literally and then you've got digital preservation and curation which is safeguarding our cultural heritage in the digital age now you might be wondering why should I care about digital Humanities why well let me ask you this have you ever wondered what Shakespeare's social network looked like or how the sentiment in Jane Austin's novel changed over time or maybe you're curious about how the language and Hip-Hop lyrics has evolved over the decades well digital Humanities gives you the tool to answer these questions and so many more so it's kind of like having a time machine and a supercomputer rolled into one but don't just take my word for it let's hear from someone who's ventured into this digital cultural Frontier as a history major I was skeptical about bringing technology into my studies but this course opened my eyes to a whole new world of possibilities and I'm now using network analysis to study the spread of ideas during the Enlightenment it's it's like being a historical detective with a really cool magnifying glass but why is digital Humanities important in today's job market well according to a 2023 report by the American Academy of Arts and Sciences employers are increasingly looking for graduates who can bridge the gap between technology and Humanities they want people who can not only crunch numbers but can also understand their cultural context in other words digital Humanities graduates are like Swiss Army knives of the job market they can code they can analyze and they can tell compelling stories with data so whether you're interested in journalism history or data analysis Anis or even Tech entrepreneurship the skills you learn in this course will give you a Unique Edge but with that being said this is one that may not directly get you a job and that's where companies like corsera come in where their certifications directly do help you to get a job and I've talked about them on this channel before I've actually done tier lists and Top 10 rankings Etc some of the professional certificates on corsera you can actually audit for free and in order to get the searchs it's usually only $40 to $50 per month so definitely check those out because the first 7 days is free regardless of what you do and I'll put those down the description in the pin comment below so click down there now let's move on to a course that's going to turn you into a modern-day Cicero get ready to master the art of persuasion and this is the art of persuasive writing and public speaking now before you run away thinking this is just about fancy words and long speeches let me stop you right there this course is about to turn you into a communication superhero and the older I get the more important that I realize communication is with all the stuff that's happening with AI there's no way that AI can make you a better Communicator that's something you just have to practice and learn on your own because rhetoric isn't just for politicians and lawyers it's a superpower that can boost your career in any field whether you're pitching an idea to your boss presenting a project to your client asking for a raise or asking your significant other to take the trash out the art of persuasion is your secret weapon so what exactly will you learn in this course well let's break it down first we got the three pillars of rhetoric ethos posos and logos which are your new best friends then you've got crafting compelling arguments because facts alone don't always cut it then you've got the power of storytelling turning dry information into captivating narratives then you've got body language and vocal techniques because how you say it is as important as what you say because one size doesn't fit all when it comes to communication I'm not a natural public speaker is this course really for me well let me let you in on a little secret most great communicators weren't born that way they learned and practice these skills and as the famous orator cisero once said the skill of speaking is not given to anyone and yet it can be acquired by everyone but don't take my word for it let's hear from someone who's been through this rhetoric boot camp before this course I was terrified of public speaking my palms would sweat my voice would shake and my mind would go blank but learning the principles of rhetoric changed everything I recently gave a presentation at work that led to a promotion and it's like I've been given a superpower but why is rhetoric so important in today's job market well according to a 2023 survey by the National Association of colleges and employers communication skills are the most sought after quality by employers ranking even higher than technical skills in many fields in other words being able to communicate effectively isn't just a nice have skill it's a musthave it's the difference between having great ideas and having great ideas that get implemented it's the secret sauce that can take your career from good to great but here's the real magic of rhetoric it's not just about speaking it's about thinking as you learn to construct your own arguments and analyze others persuasive techniques you'll find yourself becoming a more critical thinker you'll be able to see through flashy but empty arguments and craft more compelling ones of your own and the best part you can practice these skills every day every email you write every meeting you attend every conversation you have is an opportunity to apply what you learn in this course and as the philosopher Aristotle once said we are what we repeatedly do Excellence then is not an act but a habit so are you ready to become a master of persuasion remember in the world of rhetoric your words are your wand your voice is your superpower and every conversation is an opportunity to change minds and in some cases Hearts now let's dive into our final course which is going to turn you into a digital wizard get ready to enter the fascinating world of artificial intelligence welcome to the course that's going to make you feel like you're living in the future introduction to artificial intelligence with python now I know what you're thinking AI isn't that just about robots taking over the world well hold on to your neural networks because AI is so much more than that it's the technology behind your smartphone's Voice Assistant the recommendation algorithms of your favorite streaming service and even the systems that detect fraudulent transactions on your credit card in short AI is everywhere and this course is your ticket to understanding and creating it so what exactly will you learn in this course well let's break it down first of all you got Python Programming the language of AI Wizards then you've got search algorithms teaching computers to find the best Solutions then you've got knowledge representation which is how to make computers think then you've got machine learning which is giving computers the ability to learn from data and then you've got one that's very exciting which is neural networks the building blocks of deep learning and then you've got natural language processing which is helping others understand and generate human language now you might be thinking this sounds complicated do I need to be a math genius to take this course and the answer is a resounding no while some mathem matical concepts are involved the course is designed to be accessible to beginners and as the famous computer scientist Alan Turing once said we can only see a short distance ahead but we can see plenty there that needs to be done but don't just take my word for it let's hear from someone who's ventured into the AI Frontier before this course AI seemed like magic to me now I understand the principles behind it and can even create my own AI models I recently used what I learned to develop a chatbot for my company's customer service and it's been a game changer but why should you care about AI well according to a 2023 report Ai and machine learning Specialists are among the top 10 jobs with increasing demand the report predicts that by 2025 85 million jobs may be displaced by a shift in the division of labor and this division of labor is of course between humans and machines but 97 million new roles May emerge that are more adapted to this new division of labor in other words AI isn't just a cool technology it's reshaping the job market so by learning AI you're not just picking up a new skill you're future proofing your career so whether you're in healthcare Finance marketing or pretty much any other field understanding AI can give you a significant Edge but here's the real magic of AI it's not just about creating smart machines it's about solving complex problems in innovative ways as you learn to develop AI systems you'll find yourself approaching problems differently thinking more systematically and seeing patterns where others see chaos and the best part this course gives you hands-on experience you'll be coding your own AI systems from game playing agents to image recognition algorithms and as the famous computer scientist fay F Lee said if you want to make AI robust and a great technology for Humanity you've got to make it diverse so are you ready to become an AI wizard remember in the world of AI every problem is an opportunity every data set a treasure Trove and every algorithm a step toward the future so yeah these five courses are phenomenal but they may not help you directly get a job they'll teach you the underlying skills that will indirectly get you a job and they may teach you skills that will make you make way more money in the future but if you want to check out a video on training and certifications that will directly help you get a job definitely check out that video of the best corsera Sears and you can check that out by clicking right here \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_How_I_make_science_animations_da297b26.txt b/graphrag-ollama-config/input/youtube_How_I_make_science_animations_da297b26.txt new file mode 100644 index 0000000..c261d53 --- /dev/null +++ b/graphrag-ollama-config/input/youtube_How_I_make_science_animations_da297b26.txt @@ -0,0 +1,10 @@ +# Source: YOUTUBE +# Title: How I make science animations +# Channel: Artem Kirsanov +# Video ID: yaa13eehgzo +# Duration: 43 minutes +# Ingested: 2026-01-25 16:34:14 +# Word Count: 6957 + +--- +over the past few months a lot of people have asked me about the creative process behind my videos like what software I use and how some particular animations were brought to life this is why I decided to make a dedicated video where I would share with you some of my secrets and use animations from my previous videos as illustrative examples just to walk you through how they were done if you're interested stay tuned before we dive deeper into the animations themselves let me address one of the most common questions I get and that is what software do I use here's the thing unfortunately there is no ultimate tool that will help you create a video from start to finish instead every software is made for specific purposes and thus has its own limitations that's why my workflow is almost always some combination of many programs and packages that I use depending on the problem at hand so I've prepared the whole list of software that I use in video production and when exactly I use each of them starting with Adobe After Effects this is my main Workhorse that I use for the majority of simple animations as well as for composing the results of other programs into the final video to me after effects offers an optimal balance between capabilities and usability I don't really need to create any realistic explosions key in fancy color correction 3D tracking or anything like that for this purpose there are other dedicated applications much more powerful than After Effects but using them for simpler stuff would be an Overkill like trying to cut a paper with a chainsaw for example I can create the text stylize it to globe with gradient make it appear on the screen add an image that would pop up wiggle around and gradually change its Hue and finally make a smooth transition out of the scene within just a few clicks number two python now this is where things get interesting unfortunately scripting in After Effects is not as powerful as convenient as compared to some other programs yes technically there is a JavaScript API but honestly I found it to be quite unusable and there is only so much you can do without scripting create a pair of circles that would wiggle around while always being connected by a line easy but creating a hundred of such circles and lines that would form a wiggling graph with specified properties optimally positioned in a two-dimensional space is pretty much impossible this is why when I need to create something that can't really be done by hand I have to rely on other tools that would allow me to make visualizations programmatically taking advantage of mathematical functions heavy numerical calculations variables Loops recursion stuff like that typically I do everything in Python since this is the programming language I'm most familiar with and it has a few great modules for creating visualizations but hey if you have a lot of experience in other languages like C plus or Julia and would strongly prefer to use them instead there are some really great solutions for them as well the choice of the exact tool doesn't really matter anyway in my work I mostly use two python packages the first one is called many it was originally developed by Grand Sanderson also known as three blue and brown who I'm sure most of you have heard about today there is a rapidly developing version of many maintained by the community I've used it extensively in my earlier videos for all sorts of mathematical animations but I just kept bumping into things that I couldn't Implement in manim for example drawing gradient lines colored by coordinate Plus at times I found the workflow to be tedious and not really intuitive this is why about 10 months ago I gradually began to switch to another visualization module called matplotlib I'm sure most of you probably recognized that name because it's like the most popular solution for plotting or data visualization in Python what is less known however is that method lip is not limited to simple static plots like the ones you would create for a research paper in fact it has some of the most amazing animation capabilities personally I find Matlock lip to be much more intuitive compared to manim and although it is more low level in a way so that the same animation takes up more lines of code it gives me much more control and much more freedom over manipulating individual elements on the screen on a frame by frame basis so nowadays whenever I need to create an animation of a plot being drawn or visualize some complex system beat an icing model or an artificial neural network I use Matlock lab further in the video we will take a look at a few examples of how exactly it is done one thing that I still invariably do in many however is these types of graph animations this would certainly be possible to recreate in matplotlib but manim just has got such an amazing out of the box solution for graph theory that I can't ignore it within just a few lines of code it is possible to draw a graph object from Network X and make it wiggle number three blender unfortunately both After Effects and python Solutions have very limited capabilities of working with all three dimensions so whenever I need to create something in 3D B neurons mice running in mazes or fancy surface plots I use blender it is completely free and open source but that doesn't make it less powerful additionally blender has got an amazing python API which means it is possible to create some 3D visualizations programmatically as well these three pieces of software After Effects Python and blender are the backbone of my animation workflow but there are a few other programs mostly from the Adobe suite that I use every now and then at different stages of video production for example Adobe Illustrator is my go-to Vector editor when I need to draw something like a simple asset or a diagram the good thing about it is that it works seamlessly with Adobe After Effects so I can use Illustrator files to drive some animations and whenever I need to change the source in illustrator things will automatically be updated in After Effects as well Photoshop is mostly for thumbnails and minor raster work like separating a subject from a background or color correcting by the way just like illustrator Photoshop integrates nicely with After Effects Adobe Premiere Pro now although After Effects is certainly good for animating it is virtually unusable for classic video editing you know like trimming Clips arranging them in time adding sounds background music stuff like that for this type of video editing and usually at the final stage and especially when I'm shooting with a face camera I use Premiere and finally Adobe Audition for all sorts of audio work removing background noise enhancing The Voice removing plosives and other nasty things like breaths and mouth sounds and that's pretty much it all the software I use to make my videos now let's be more specific and take a detailed look at how some of the animations were done here's the list of what I'm going to talk about along with time codes so you can easily find a particular animation you're looking for or you know just watch the whole thing that would be awesome as well were wondering how mad but lib can be used to create mathematical animations over the years I've developed a few tricks and strategies on how to use map.lib in Synergy with After Effects and this is actually the key takeaway no single mathematical animation you can find in my videos was created with mat.lib or manim alone there is always some After Effects involved to enhance the animations I've actually prepared a short animated clip to use as an example don't search for any deep meaning in it essentially the only purpose of this toy animation is to illustrate various approaches here it is consider an arrow rotating around a circle with variable speed if we trace the y-coordinate of the arrow tip we will get a sinusoid with time varying frequency let's say we want the amplitude of the sinusoid to change in time according to this function right here let's zoom into the resulting wave and make it wiggle for a while between the two states just for fun alright great now let's break it down piece by piece first I usually identify the core components the building blocks of the scene for this first portion right here these would be the arrow spinning around the Rainbow Circle the graph of a wave colored by the face being gradually drawn along with the field graph of frequency as a function of time these three elements should be synchronized to each other and be optimally arranged on the screen along with some texts once I've identified what exactly needs to be done it's time to determine the tool that I'm going to use for each of the jobs in this case either after effects or python now of course there is no right or wrong answer since it's the matter of personal preferences and experiences for me draw in such a wave from scratch in After Effects and synchronizing it to the spinning Arrow would be a pain so I'm better off creating something like that and inapt.lib on the other hand arranging everything on the screen and animating texts directly in Python would be tedious if not impossible so let's leave that to After Effects once individual jobs are allocated to their respective software it's time to get to work alright so in Python let's start off by creating an array of instantaneous frequencies that would tell the arrow how fast it should move at every point in time and the array containing the resulting sinusoid for visualization let's set up a method lib axis with high enough resolution make it completely black and add a nice thin grid to plot our array as a gradient line colored by face values we can use metal Clips align collection object specify the array of colors obtained from the array of phases by passing it through a color map stylize the line a little bit and finally add the resulting line collection object to the axis voila we got an image of a wave but we need it to be animated in for this purpose let's use the funk animation class available in the animation submodule of mat.lib essentially actually the way it works is you define a function that would be repeatedly called and would modify the plot at every frame of the animation this function let's call it animate wave we'll accept the parameter specifying the current frame for this case since we are creating a drawing animation let's call this parameter T current and we are going to animate it from the first value of our time array which is 0 to the last which in this case is 5. whenever the animate wave function is called it should hide the portion of the graph where time is greater than T current and show the portion where the time is less than or equal to T current the way we can do it is the following remember when creating the line collection object we specified the color for each point well in a similar fashion it is possible to specify the opacity of each point by the way in Python opacity is usually called Alpha so we can call line collection dot set Alpha the expression in the parenthesis will compare each element of the time array to the value of T current and return a binary mask of ones and zeros which we can use to set the opacity of individual line segments to create the animation let's create an instance of the funk animation class passing our figure object the animation function and the list of frames in this case we are going to animate the value of T current from T start to T end and let's make it 5000 frames to make the animation smooth let's tell python that the interval between consecutive frames should be around 30 milliseconds so the animation will be saved with 30 frames per second finally we need to call Dot save on the resulting animation object and a few moments later we got a nice animation of a wave being drawn I think it's a great point to pause and talk about the number of frames you probably noticed that the resulting video is extremely long two and a half minutes and the reason for that is because I need the animations to be synchronized with my voice I want to easily control the speed with which they play but here's the thing if the animation is long I can easily make it 5 times faster and it will look nice I can even tweak the rate and make it non-linear to achieve the easing effect but going the other way around making the fast animation 5 times slower although technically possible will give you a very choppy and ugly result in the first case the file contains a large number of frames let's say 5000 and to speed the animation up the video editing software simply takes every fifth frame change them together and the result looks good alternatively to slow a faster animation down it needs to somehow stretch the existing thousand frames into 5000 since it can't augment the pre-rendered video with any new frames the existing ones are simply repeated so that now you perceive the animation as if it is played with 6 frames per second instead of the normal 30. this is why when creating the animation building blocks in Python I intentionally save them as videos with humongous number of frames because I can always easily throw some of those out but to generate new ones would require rewriting the entire code similarly to how when you are buying wallpaper rolls it's certainly better to overestimate than to underestimate OK let's get back to matplotlib in exactly the same way we can animate the frequency graph the only difference is that it is completely white instead of the gradient and also includes a fill let's create the fill object using x.fill between method and modify our existing animation function a little bit now on every frame along with setting the opacity of the line segments it should also modify the fill as of right now matpatlib can't modify the existing polygon but we can easily just delete the old fill and recreate the new one with necessary limits on every frame specified by T current for the arrow let's set up a black figure with polar axes and specify how ticks and the grid should look like we can draw the colored Circle in exactly the same way we did with the way just generate an array of angle values linearly spaced from 0 to 2 pi along with the array of radii all the elements of which will be equal to 1. in order to add the circle to the axis we can use the code with a line collection from before only now for the case of polar axis segments will be specified with angle and radius values instead of X and Y to create an arrow let's define a function called get polar arrow that will take the value of the angle and add the arrow to the axis by calling x dot Arrow rotated and colored according to the angle to create the animation itself we need a very simple animation function that will remove the old arrow and create a new one on every frame as we gradually change the angle according to our array of phases and here we go next in the animation there is this graph of the amplitude being drawn together with the field and later the copy of the curve should detach from The Fill fly down and kind of Squish the sinusoid into the target shape this part is a bit tricky and there are always many ways you can achieve the same result what I suggest we do is the following in Python using similar approaches first animate only the amplitude curve without the fill you'll see why in a minute then by modifying the code slightly animate only the fill and save the result into its own video file unfortunately matplotlib can't really color these fields with gradient so we'll have to tackle this on post processing for now let's make the fill fully white finally before we can compose everything we need this last animation of one wave being gradually transformed into another took me a while to figure out how to do this but this solution turned out to be pretty straightforward first create the line collection corresponding to the initial stage as we did in the beginning the animation function should gradually interpolate between the curves the initial array wave and the wave times amplitude array to achieve this the function will take a single parameter called proportion which is a number between 0 and 1 specifying where we are in the interpolation process so on every frame the animation function will mix the two arrays accordingly and change the segments of the line collection object as you can see when the proportion is equal to zero this expression evaluates to just wave while when the proportion is 1 it equals to wave times amplitude and everything in between let's animate the proportion value from 0 to 1 say with 500 frames and boom smooth interpolation between the two functions pretty neat right believe it or not but now we have all the necessary building blocks we need to create the full animation let's put python aside for now and finally move to Adobe After Effects now I'm not going to explain every single thing just cover the key ideas in After Effects I usually scale and position the layers accordingly to compose the scene by the way notice that these video files have black background which looks kind of ugly if you want to put them on a nice dark but not completely black background in addition this prevents them from overlapping to solve this issue pretty much for every single asset I set the blending mode to screen I'm not going to explain the theory behind blending mounts if you're interested check out this great article on Wiki essentially everything that's black will be made transparent so that now not only can we see the background but it is also possible to arrange them however we want without one layer obstructing another we can now adjust the speed of the animations using the time remapping property and arrange them on the timeline for instance I want the amplitude animation to appear only after the first three animations relating to the arrow are Dawn plane now because the amplitude curve and the fill are two separate video files we can take advantage of that and do something about this horrible white fill in After Effects using the gradient ramp we will fill the layer with orange and red gradient according to the curve and set the layer mask to its own copy so that now only the pixels that were white in the original file will be shown while the other pixels that were black will be transparent now just lower the opacity and voila a nice gradient fill that is animated together with the Curve now to the squishing part notice how the last frame of the video with the wave being drawn is identical to the first frame of the video with the interpolation between the waves because of this we can seamlessly Stitch the two videos together as long as both layers in After Effects have the same position and scale foreign as a result the wave is first being drawn and then after a pause is being morphed into the amplitude modulated version of itself while the rate of both animations as well as the duration of the pause can be easily tweaked in After Effects by animating the time remapping property to match the voice as a nice touch we can make a copy of the amplitude curve layer and animate its position and opacity to achieve this effect as if the curve is kind of squished in the wave in order to zoom into the wave let's parent all the layers to the wave interpolation video so that they will inherit the Transformations and animate the scale and the position of the parent layer and maybe simultaneously animate the opacity of some of the child layers to make them Disappear Completely finally as you may have guessed already we are going to use the time remapping property to create this wiggling animation this is where the fact that the source video contains a large number of frames will come in handy and and that's pretty much it just add the texts maybe stylize the animations by adding a few effects to your liking and render the video all right so just a quick recap of how Python and after effects can be used in tandem matplotlib is used to create the mathematical building blocks of animations those building blocks like graphs or array images usually are then scaled arranged and composed together in After Effects which also helps with synchronizing the resulting animation to the voiceover and remember time remapping is your best friend well I hope this was helpful and with a major block of mathematical animations out of the way let's look at a few more tricky ones foreign what you see right here is the biophysically accurate description of how membrane voltage propagates through a pyramidal neuron during action potential let me tell you a short back story of how such animations were born in the first place while I Was preparing to make a video on dendritic computations this one right here I realized that I needed some way to animate realistic dynamics of membrane voltage in space and time since you know this was the central point of the video however after searching the internet for a solution I just couldn't really find anything that would work so I did what every rational person would do in this case I created my own tool the main idea was to run biophysical simulations using real neuron morphologies in the free neuron simulator environment and then somehow brain simulation data into blender along with geometry this task turned out to be not as straightforward as I hoped for and it took me a couple of weeks of experimenting and going through the blender API documentation before I could make it work realizing that I will probably need this for future videos and other people might find that helpful I went ahead and created an actual blender add-on called blender Spike which is now available on my GitHub along with the detailed instructions now you may be wondering why not name It Blend or neuron because that would be so much cooler turns out there already is an add-on called blender neuron developed in 2018 designed to do exactly that but unfortunately no matter how hard I tried I couldn't make it work anyway creating your own simulations from scratch in blender Spike requires knowledge of Python and the basics of neuron simulator to load the morphology set up the biophysics and run the simulation results are then exported into a blender friendly format with a little companion module called blender Spike Pi this resulting dot pickle file essentially contains all the data including the morphology of the branches and the frame by frame voltage data for each branch the good news is once you have the dot pickle file with the simulation results for example by downloading an existing one from my GitHub or asking your simulation proficient friends to create one for you you can simply dump that into blender and easily customize the appearance of the neuron the color map glow intensity to build your own unique animations to take it one step further it is possible to combine blender Spike with matplotlib animations which I discussed previously it is straightforward since blender updates the voltage by looking up values from the python array stored in the pickle so for example we can render the neuron in blender animate voltage graphs in matplotlib and compose and sync the two videos in After Effects for a more complex animation another animation a lot of people are interested in is this slicing through probability distribution from the video on cognitive maps not gonna lie this is one of my favorites as well this was created in blender with just a little bit of python by the way the exact method I'm about to describe was used for this animation from the wavelet transform video as well now the first step is to construct the three-dimensional surface inside the blender unfortunately there is no native way to just tell blender to plot a surface so we will have to use a workaround we will first use Python to generate a black and white image called the height map which means that white pixels correspond to more elevated areas this can be done with net.lib by calling the dot emcee function with a binary color map here is what the resultant image looks like in blender we can now create a grid object apply the displace modifier and specify the texture to be the displacement map we have just saved what this will do is extrude the vertices of the grid as specified by the brightness of the image and voila a nicely looking 3D visualization of the array one way to cover this would be to create a color image in matpatlab in a similar Manner and then apply that image as a texture to the object but in this particular example since the coloring is quite simple namely it's just a gradient along one axis we can create it right inside blender to specify materials we are going to use blenders node based Shader editor essentially it allows to create complex materials by routing basic computations and materials through a system of interconnected nodes for example we can modify the default green material by specifying that the base color should be taken from a gradient using a color map node and that the position along the gradient should depend on the y coordinate you can now manually create any gradient you want or use this tiny add-on called blender color maps to quickly bring gradients from MacBook lip color Maps into blender alright now to the slicing part the key idea behind this is that inside the node editor blender allows you to mix different shaders in different proportions which could be a function of variables including other objects sounds confusing but here's what I mean suppose I want to mix the gradient Shader with a fully transparent Shader well I can just add a mixed Shader node in blender and change the factor slider which is the proportion in which the two shaders are mixed so for zero only the gradient is visible for one the object is fully transparent 0.5 somewhere in the middle you get the idea the cool thing about it is that this factor is not restricted to being a constant value for example we can plug the Z coordinate of the object there to achieve this cool fading effect which depends on the height in order to slice the surface let's create another object an empty plane it will not be rendered and will use it only to drive the material namely let's take the y-coordinate of the empty object threshold it with some value and feed the output into the factor of the mixed Shader that way the vertices of our surface will have either one or the other Shader depending on the location of the empty object foreign to achieve this thin white line and the boundary the idea is similar we just create a third Shader that will be a pure white glow and mix the three shaders depending on the position of the cutting plane there is just this funny note set up to work around the limitation that the blender can't mix three shaders simultaneously so I first have to mix the two shaders together and then mix the result with the third now we can duplicate the surface object apply the wireframe modifier to one copy well to make it wireframe and simply reverse the order in which the shaders are mixed so that now the wireframe is visible when the original surface is transparent and vice versa and now what's left to do is the animate the movement of the cutting plane make the camera spin around and here we go foreign about the three-dimensional animations of brain structures the models themselves come from the existing brain atlases in particular the ones published by Alan Institute downloading them can be a bit of a challenge given that the interface is not really intuitive I found that the most convenient way to use brain atlases is through brain Globe API which provides a python interface to download and navigate the data then what's left to do is navigate to the folder where the atlas is stored locate the necessary.obj file since they are named by their IDs and bring the model into blender but this can be tedious especially when you want to bring multiple brain structures into a single scene to simplify the process I've put together a tiny blender add-on called blenderbrain very original name in I know which you can find on GitHub it allows you to import meshes from specified Atlas in one click simply by specifying their acronym which you can look up in the corresponding structures.csv file that the blender Globe downloads for example let's say I want to look at the ce3 region of the mouse keeper campus well I can just select the atlas and type in ca3 similarly I can bring the dented gyrus into the scene by typing DG notice how it automatically gets positioned into the anatomically correct place and suppose we want to look at where these two structures are located relative to the entire brain typing in Gray Imports the entire gray matter and since hippocampus and Dente gyrus are subcortical structures we can't really see them now let's change that make the gray matter almost transparent change the environment settings so that there is some backlight and color of the brain structures for instance let's make the dentage iris glow with blue and add this subtle gradient to the glow of the C3 region along the x-axis what's left to do is to animate the camera to spin around the brain render it with a black background and the footage is ready to be used in After Effects finally I have prepared something special for you namely let's explore how to build this animation of information transmission from the brain criticality video yes including this segment where the neurons are being rearranged now I realized that this section will be a bit more coding heavy and I will skip through some of the technical details so please be prepared for that by the way you can find all the code for this video on my GitHub as well okay so just to remind you we want to animate the simulation of a so-called branching model it consists of M layers and each layer has n neurons that can be in one of two states on or off layers are connected sequentially and each connection has a certain transmission probability associated with it so that information can spread from left to right this number Sigma which is equal to the average number of neurons activated by one Downstream neuron controls the the behavior of our Network additionally each neuron has a very small probability of being spontaneously activated even when it doesn't receive any input before animating the activity it's necessary to set up the simulation itself let's define a function that initializes the N by m array of zeros that is going to store the state of the network next we create a function called Network advance that will advance the network one step into the future and return the update state to advance a network we need to First randomly activate a small subset of neurons in order to model this stochastic input and model the propagation of information as specified by the value of Sigma once we have a function that advances one time step we can just call it sequentially a few hundred times to get the full simulation result great so now we have a stack of 2D arrays containing the entire evolution of the network now we just need to somehow animate this because each frame should depict one state of the network which in turn is given by a two-dimensional array a natural solution is to use methodly functions like IM show or P color mesh then on every frame we just need to change which of the arrays is being plotted so just like before set up the figure and axis plot the network state I'm going to use P color mesh here Define the animation function that would change the data depending on the current frame and call func animation to create a video however there are a couple of crucial problems with this animation first of all it is too fast so much that it is literally painful to look at it this is because we currently change the array on every single frame and the interval between the frames is short we can try to increase the interval but now the animation is awfully topic let's do something else instead consider the Dynamics of a single neuron in isolation right now if we plot its state as a function of time we will get something like this zeros interspersed by a few ones which makes sense we are going to cheat a little bit and kind of smooth out its activity in time so instead of instantaneously jumping to one and then back to zero the neuron State Should gradually increase and then decrease according to this shape right here which consists of two exponents so we need to replace every single sudden jump with this gradual rise and decay in mathematical terms we have to convolve the activity function with this exponential thingy called the kernel this operation can be easily done in Python using numpies can evolve 1D finally in P color mesh we can choose a nice color map and animate the smoothed States array instead just play around with the time spread and the kernel shape to achieve the optimal balance as a side note if you don't like the squares that the peak color mesh produces with just a little bit of After Effects they can be changed to pretty much any shape for example circles just create a shape layer on top of the video draw one circle of the necessary size add the repeater with a proper number of copies and tweak the spacing to ensure that all the circles in the row fall on top of the squares for a one-to-one mapping add another repeater and do the same for the vertical spacing now just change the track matte of the video to Alpha matte essentially the video with our colored squares will be now masked by the circles and also feel free to add the glow finally let's create this rearranging animation right here which by the way if you are into manim can be quite useful the code for running the simulations and smoothing is the same but this time instead of map.libs color mesh we are going to use the graph object inside of menu the main idea is to run the network simulation but during the animation initially Shuffle the position of individual neurons on the screen then animate how each neuron gradually returns to its original place in the multi-layered network I know it's a bit backwards and kind of looks like cheating but hey it gets the job done so let's create a network of 10 layers with 10 neurons in each layer and run it for a couple thousand iterations then use this function that I stole from Network X documentation which creates a multi-layered network X graph object from this specified layer sizes in our case that's 10 layers with 10 nodes each I've actually modified it at a little bit so that now not all the edges are shown only a random subset of them to make the animation not too crowded with the lines inside the definition of a Manning scene let's create two coordinate systems one for the square grid and the other one for the layered layout that will map the index of each neuron to its position on the screen then randomly Shuffle the positions of all individual neurons and create a mapping a kind of a lookup table specifying that the first neuron in the first layer should be located here second neuron here Etc we can now create the graph object and tell Magnum its layout the positions of individual nodes to animate the colors of the nodes according to the simulation data we are going to make use of the magnum's value tracker class let's create an updater function that will be called on every single frame this updater will change the colors of our graphs nodes according to a color map and the value that's been passed into the color map is obtained by taking the value of our value tracker on a given frame and interpolating the simulation array for every neuron finally we need to connect the updater function to the graph object and animate the value tracker to make it transverse a specified number of frames this will create the animation similar to the one we had in Matlab but with the positions of neurons being randomly shuffled but what if after playing for a few seconds of activity in this shuffled State we want to rearrange the neurons into the original layered structure without interrupting the activity animation well luckily manium is very clever about playing multiple animations together so all we need to do is create a list of animations that will move each node into its original position append this list with an animation for incrementing the value of the value tracker and play the resulting collection of animations lastly let's play a few seconds of activity in the layered configuration by incrementing the value tracker without moving the nodes by the way I hope you can appreciate how convenient it is to use manim's graph object since when we animated the positions of nodes all the edges were animated automatically as you can see pretty much every single animation we covered today involved python in one way or another indeed this programming language is quite powerful and has enormous applications if you'd like to get started with python and potentially create your own visualizations you are going to love our today's sponsor brilliant.org brilliant is an exceptional online platform that offers interactive courses in stem Fields what sets brilliant apart is their emphasis on Active Learning and problem solving approach enabling you to tackle real world challenges with confidence the courses are packed with interactivity and stunning visualizations that help you develop an intuitive understanding of even the most challenging Concepts while problems and quizzes further consolidate the knowledge brilliant offers over 90 courses on a variety of subjects for multiple levels of difficulty whether you are a complete beginner or a professional looking to expand your skills brilliant has got you covered for example if you are interested in implementing some of the visuals from this video you may want to check out their course called programming with python to get a firm grasp on the fundamentals such as variables loops and functions and from there you can move on to algorithm fundamentals to learn more complex algorithms like array sorting and stable matching don't hesitate to take curiosity to the next level and start learning at your own pace just by dedicating 15 minutes a day go to brilliant.org artem care sonoff to get a 30-day free trial of everything brieland has to offer and the first 200 people to use this link will get 20 off the premium subscription all right well I hope some of it was helpful if I left out a particular animation and you would like to know more about it let me know down in the comment section and who knows maybe I'll create a part too in the meantime if you enjoyed this video press the like button share it with your friends and colleagues And subscribe to the channel if you haven't already stay tuned for more interesting topics coming up goodbye and thank you for the interest in science visualization \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_How_to_Use_Excel_to_Create_a_Project_Management_Da_b96c18e9.txt b/graphrag-ollama-config/input/youtube_How_to_Use_Excel_to_Create_a_Project_Management_Da_b96c18e9.txt new file mode 100644 index 0000000..8b83315 --- /dev/null +++ b/graphrag-ollama-config/input/youtube_How_to_Use_Excel_to_Create_a_Project_Management_Da_b96c18e9.txt @@ -0,0 +1,10 @@ +# Source: YOUTUBE +# Title: How to Use Excel to Create a Project Management Dashboard +# Channel: MyExcelOnline.com +# Video ID: Yi9Rse-y22M +# Duration: 14 minutes +# Ingested: 2026-01-18 13:47:21 +# Word Count: 2700 + +--- +- Hi, and welcome back to myexcelonline.com. Today we are going to show you how to use Excel to make a project management dashboard. If you want to learn more about Microsoft Excel and Office join our academy online course and access more than a thousand video training tutorials so that you can advance your level and get the promotions, pay raises or new jobs. The link to join our academy online course is in the description. So here I am in Excel and I've created a sample project schedule where I'm going to make a few different series of videos. And there are a few tasks that are associated with every video. For each of these I have a start time, a duration, and a due date. And you can also see in this column I the progress of how complete my task is. I'm also going to include the amount of budgeted money and the actual money spent so I can get an actual divided by budget percent over here. So this is great information for anyone that's managing a project, but it would be really nice to have a dashboard that updates really quickly so that we can see those metrics all on one screen. So how do we go about doing that? Well, first of all, I'm going to rename sheet one to data. I'm just going to double click here and type in data. And then I'm going to hit this ad button right here and I'm going to double click and call this one dashboard. Now, the first thing I would like to do is maybe click and sell a five on my dashboard and go up to insert and go to pivot table. And I'm going to do that from table range and I'm going to click this up arrow and go over here to data. And I'm just going to select all the data that I have in my table and hit this button again and say, okay so now I'm going to be asked for which fields I would like in my pivot table. Well, I'm going to select the series as well as the video and the tasks right here, so that I can filter this data out. For my values, I would like budget and actual. And for my columns, I'm going to leave this summation of values and then I'm going to close this. So if I look over here, I can see my different series. I have a dashboard, a power query and then I have my tips series and I can see the sum of my budget and the sum of how much I've spent on my actuals right here. And then I can expand each of these and see where the differences are. I can also go here and close up all of these so that I can see just my list of videos underneath each main category that I have right here. So as you can see I can quickly get to the information that I wanna see. So what if I wanted to filter out all of my data that I have here? Well, if I select my pivot table right here and go up to home and go up to insert and go over here to Slicer and let's try to do this by video and say, okay so now I can see that the Slicer has created a list of projects for me. So if I select just one project I can see that my pivot table is updating. And if I expand this, I can see each of the roles and how much the budget is and how much the actual is. And if I wanna select more than one video right here I can click this button and I can select more than one. So I can see that the finance video this is the amount of money allocated for that one. And the intro video to Power Query this is the money that's been allocated to that one. So in the case of finance I'm actually coming in under budget but in the case of Power Query, I'm actually at my budget and overall my actual costs will be lower than my budget. So let's make A one a little bit bigger shrink our slicer a little bit and just move that over here. If you are liking this video, please give us a thumbs up and subscribe to our channel and hit the bell button to get notified when we release our weekly videos. Now let's say I would like to see other information about my video. So next, let's insert a second pivot table. I'm going to go up here to insert pivot table from table range. Again, I'm going to click the up arrow go over here to data select my whole table click this button again and say, okay. This time I'm going to choose video, tasks, the duration the days complete and the progress. But if I look here there's a problem because my duration's being summed up and I really don't want these sums appearing here. So what I'm going to do instead is drag each of these over here to my rows. But if I look here, this doesn't look very neat. So what I would like to do is click here, go up to design go over here to report layout and just say show in tabular form. So now I can see my progress over here as a percentage. So this is looking a little bit better but I would also like to go back to design and go to subtotals and say, do not show subtotals. And then also the buttons are kind of in the way taking up space. So if I go back to pivot table analyze I can turn those off right here by clicking the buttons. So now this table is making a little bit more sense. I can see how long it's supposed to take how many days are complete and what my progress is. And if you would like, you can hit the X right here to have a little bit more space on your screen. So now I'm going to highlight the progress and go up here to home and go over here to conditional formatting. And let's put some data bars in here so we can get a little bit better visual on the completion rate of each of our projects. And let's scroll up again and right click on the slicer and go down here to report connections. And I can see that the slicer is linked to the first pivot table but I actually want it to be linked to both pivot tables so that they change anytime I select something different in the slicer. So I'm going to put a check on both of these and say, okay so now when I pick all of these I can see that if I scroll down every project is included in both of my pivot tables. But if I deselect a few of these I can see that my pivot tables are changing to reflect what I have selected in my slicer. And at this point, I would like to narrow down my selections for my pivot tables a little bit more. So I'm going to go over to insert and go to Slicer and I'm going to pick tasks and say, okay and I've now been given a series of tasks that make up each one of my videos. So if I right click on here and go to report connections I wanna make sure I pick both pivot tables so that they will be changed as I change data in the slicer and say, okay. And then if I wanna be able to multi-select I need to select this button right here. Let's just say I want to pick the finance video. And over here I just wanna see how my progress is coming on the editing. So I can now see that these are my budget numbers and this is the time that I have that's gone into the editing. Now let's go back to my data. What other information on here would be useful? What I'm thinking is how about anything that has a progress of zero has not been started? Anything with the progress of a hundred is finished and anything in between is something in progress. So let's create that little table of information right here, complete in progress not started. And if I scroll over just a bit and click in the cell next to complete, I'm going to type in equals and do a count if and put a left parenthesis. And I'm going to do this range and then I'm going to say comma one close parenthesis and hit enter. And I can see that there's two that are a hundred percent complete for the, not started, I'm going to put an equal and do count if left parenthesis highlight the progress column and do a comma zero close the parenthesis and hit enter. And I can see that six have not even been started. Now the tricky one is right here. And for this one I'm going to do equals count ifs left parenthesis First I can think of it being the progress is greater than zero. And second, I can think of the progress being less than one. So I'm going to highlight this column first as my first range, and then I'm going to put a comma and then say greater than zero, and then put another comma. Now I can see that I'm being prompted for a second range. I'm actually just going to select the same range again and do a comma and do less than one close my parenthesis and hit enter. And I'm going to get an error. I'm going to say, okay to this error because the greater than zero needs to be surrounded by double quotes. And the less than one also needs to be surrounded by double quotes. So now if I hit enter, I can see that this calculated to 12. So let's just look at the sum here is 20 and I also have 20 rows in my table. Well, actually I have 21 but I need to subtract the header row up here. So this is correct. So I'm going to select my data here and go up to Insert and go to recommended charts. And I'm going to select all charts and go down here to bar and go over here to this stacked bar. And when I go to Stacked Bar I can see I have three different options. The first one shows me everything as a bar. The second one shows me the in progress and then not started. But what I'm really interested in is the third one right here which shows me complete in progress and not started. So I'm going to select this one and say, okay and I can see this chart right here has been created. So I'm going to hit control X on that chart go back to my dashboard and hit control V. And let's move this chart over here and maybe make it a little bit smaller and we can change the title here to Project Status. And I'm going to close this and see that my chart is here. And this is not linked to my slicers right here because the source of this chart is not coming from one of these pivot tables. It's coming from the data tab and the little table that I created here. However, if I go in here and I change the days complete to one instead of two here, the recording on this one as well, bringing that one up to a hundred. And then let's just make one more, a hundred percent. Just pick this one right here and put a two. So now I can see I have four complete ones. Going back to my dashboard that blue has definitely grown up to 20% because four over the 20 that we have is one fifth, which is 20%. So this is correct. Going to move this over here, and let's go back here and make one more little pivot table here that we want to base a graph on but not have it show up in the dashboard. So I'm going to go to insert pivot table from table range select the up arrow, select my table click this button again and say, okay and I'm just going to take actual over budget here and I'm going to do value fields settings and just go to average. So I can get an idea of the average actual over budget percent that I'm getting and say, okay, and close this. And I can see I'm actually doing pretty well. It's a little over a hundred, but not too much. So if I click on this and I go up to insert and go over here to a pie chart let's just do a 3D pie 'cause it'll look better. And then right click on here, let's go to add data labels. And if I click on this label, I can move it to the center and let's right click and go to format data label. And let's go to text options. Say my text fill instead of black, let's make it white. And let's go back to label options scroll down to number, and instead of general let's make this percent and close that. And then I'm going to highlight the legend over here and hit delete so that it will center my graph a little bit more. So even after that, the label's still a little bit small so I'm going to double click on that until it's all highlighted. Go up to home, go to the font number right here and let's make that, let's see, 18 and say, okay. And then up here for the title let's pick average actual over budget. And then I can right click on this and go to cut go back to my dashboard, right click and go to paste. Make this a little bit smaller. So now I'm starting to get some real metrics on my projects here. Again, I can select all my projects and see all my information. I can select all of my tasks and see everything. I can also narrow it down and select fewer things and I can see my overall actual over budget. Let's click on the label here and make it a little bit bigger again, if I select it once and then click again, so I get all of these dots around it, I can click and drag to make that a little bit bigger and I can also see my project status. So as you can see, there's a lot you can do to create a project management dashboard. There are lots of other kinds of graphs. It depends on how you wanna look at your data. But I just wanted to show you today that is very doable to create a project management dashboard that you can change with just a couple of clicks inside of Excel. If you have any questions or comments, please leave them below and we'll get back to you. Thanks for watching and see you again next time. If you want to learn more about Microsoft Excel and Office join our academy online course and access more than a thousand video training tutorials so that you can advance your level and get the promotions, pay raises, or new jobs. The link to join our academy online course is in the description. - If you like these videos, subscribe to our YouTube channel and if you're really serious about advancing your Microsoft Excel skills so you can stand up from the crowd and get the jobs, promotions and pay risers that you deserve, then click up here and join our Academy online course today. \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_I_just_replaced_myself_with_Clawdbot_heres_how_3b424534.txt b/graphrag-ollama-config/input/youtube_I_just_replaced_myself_with_Clawdbot_heres_how_3b424534.txt new file mode 100644 index 0000000..04b907a --- /dev/null +++ b/graphrag-ollama-config/input/youtube_I_just_replaced_myself_with_Clawdbot_heres_how_3b424534.txt @@ -0,0 +1,10 @@ +# Source: YOUTUBE +# Title: I just replaced myself with Clawdbot… here's how +# Channel: David Ondrej +# Video ID: 2zWGFLrTVmI +# Duration: 20 minutes +# Ingested: 2026-01-26 19:50:09 +# Word Count: 3961 + +--- +Claudebot is taking over the world. It's a self-hosted AI assistant that you can message from any app and it's now even more popular than Cloud Code itself. So, in this video, I'll show you how to set up CloudBot, how to connect it to WhatsApp, as well as some of the most insane use cases. And no, you don't need to buy a Mac Mini to use it. Now, this might be the single most insane use case I've seen of CloudB showing its power. This comes from Alex Finn and what he did is like it'll blow your mind. Okay, so just to see what would happen, I texted Henry, my clawbot, to make a reservation for me next Saturday at a restaurant. Pretty standard stuff so far. When the Open Table reservation system didn't work, it used its 11 lap skill to call the restaurant and complete the reservation. AGI is here and 99% of people have no clue. So, Claudebot decided to use one of its skills to generate AI voice with 11 Labs and call the restaurant to complete the booking when the booking portal didn't work. Wild. Another incredible use case that I've seen comes from Aaron. Cloudbot can manage multiple different AI agents. So for him, it takes an idea, manages Codex and Claude, debates them on the reviews autonomously and lets him know when everything is done. So you can deploy a whole feature with multiple different AI agents debating and even the PR gets merged because Cloudbot acts as the manager as the agent orchestrator of Cloud Code Codex or any other AI agent you want. So what actually is Cloudbot? Well, as the website says, it's the AI that actually does things. It's a powerful AI assistant that lives in its own operating system. It can clear your inbox, send emails, manage your calendar, check you in for the flights, and the best part is it does it from apps that you already use such as WhatsApp, Telegram, or any other chat app that you already have. And the setup is super simple as well. And of course, I'm going to go into more detail later in the video and show you everything step by step. And no, you don't have to be a programmer to use it. Here's more information on what Cloudboard actually does. It runs on your own machine, so you don't delegate your data to Enthropic or OpenAI. It can use any chat app, WhatsApp, Telegram, Discord, Slack, iMessage, whatever you use. It has persistent memory that's stored on the machine. It has browser control. It has full system access. And it works with skills and plugins. So, it's very extensible. It's very improvable. And as I mentioned already, it works with everything. Now, let me show you just how easy it is to set up Cloudbot. Okay, so the simplest and easiest way and the most affordable way to use Cloudbot is with your own VPS, your own virtual private server. because otherwise you would have to spend hundreds of dollars buying a Mac Studio, which you don't have to do. So, I'm going to show you how to use it with a VPS to access the terminal. And honestly, this is probably the most difficult 60 seconds of this video. So, if you can just log in for the next minute and pay attention, you're going to make it through. So, on the website, which is cloth.bot, and again, I'm going to link this below. We scroll down to the oneliner command. And this is beautiful command because it installs everything we need, both NodeJS and Cloudbot. So, I'm just going to copy this. Boom. go into my hostinger panel and I'm going to access the terminal in the top right. This opens the SSH to the terminal and I just paste in the command here and hit enter. This is the cloud installer. It installs u it detects that you have Linux. It checks if you have NodeJS. It checks if you have git and then if you already have these it's good. If not it installs them automatically and then it installs Cloudbot the latest version of Cloudbot. Now while this is installing Cloudbot let me show you some of the most insane use cases that people have found. So Dan uses CloudBot for time blocking its tasks in his calendar based on importance. Cloudboard also leads him through a weekly review based on all the transcripts from his meetings and notes he jotted down in his notion. It also notifies his wife and his son about upcoming school tests. Researches big projects breaks them down into small tasks. Researches people before meeting them and creates briefing docks. Spawns background sub agents to research business ideas when he says idea. Manages the calendar for any conflict autonomously. Creates invoices and summarizes work beautifully. And this is just a few of the things Cloudbot can do. So when people say that this might be the closest thing to AGI, they are not exaggerating. And again, in this video, I'm going to show you everything you need to know to set up Cloudbot on your own VPS. So if you actually watch this video until the end, you will have one of the most powerful AI tools ever created hosted on VPS. And you can start delegating some of these tasks to your very own CloudBot. And let's go back to our terminal. And there it is. Cloudbot has been installed. Now it gives you a security warning because it is a very powerful agent. It can do anything in your operating system, which is exactly why you need to set it up on a VPS or a dedicated computer. Definitely don't put cloudboard on your main machine because again, it can delete files. It can mess stuff up. So that's why I would highly recommend you put it on a VPS. Now it says, I understand this is powerful and inherently risky. Continue. Yes, continue. Onboarding mode, quick start on manual. Let's do quick start. We need to select the model provider. Now, if you don't have your own VPS, then you need to set up one. And again, this is way more affordable than buying a Mac Mini, which the cheapest one goes for like $600. While you can get a VPS for like seven bucks a month. So, what I do for all of my VPS servers is I use Hostinger because it's not only the most simple to set up, it's also one of the most affordable options out there. Plus, they have a pretty crazy New Year sale, which I would definitely recommend you take advantage of. Now, personally, I have the KVM2 plan, which starts at just seven bucks a month, and it gives you two vCPU cores, 8 gigs of RAM, 100 GB of disk space, and 8 TB of bandwidth. So, just click on choose plan, and this will take you to the Hostinger card. Now, here I would recommend you select the 24 month plan to get the best deal possible. Now, if you want an even better deal, just scroll down a bit, go to the right, and click on have a coupon code and type in code David for another 10% off. Then, go to the left, select your server, whatever is closest to you. operating system here. Select plain OS and just click on Ubuntu latest version. Confirm. This has the most tutorials, most support, and it's perfect for cloudbot. Once you click that, scroll back up and click on continue, which will take you to the checkout page. And all that remains is just filling out your first name, last name, and create card details to buy your own VPS. Now, once you purchase, you'll get redirected to the Hostinger panel where it takes like 5 minutes for your VPS to boot up and then you can resume setting up your Cloudbot. Now, once you're in the Hostinger panel, just go to the left, click on VPS, find the VPS you just purchased, click on manage, and then in the top right, you'll see the terminal, which will take you to your terminal where you can continue setting up Cloudbot. And this is why I use Hostinger for all of my VPS's. It's just super easy to set up, and you don't need to spend hundreds of dollars on your own Mac Studio. So, thank you Hostinger for sponsoring this video. The link to Hostinger will be below the video. All right, so after this finishes running, it gives you a security warning because again, this runs on a machine. That's why we're setting it up on a dedicated VPS. You do not want to run CloudBot on your main computer. Okay, do not do it. It's way too risky. It Yeah, it breaks all security principles. So, you either need to spend hundreds of dollars on a dedicated computer like a Mac Mini or you just get a hostinger VPS. So, then you need to acknowledge that this is powerful and risky. Click on yes, onboarding mode, select quick start, and then it will ask us what provider we want to use for our AI models. All right. Now, obviously you can use local models with like Olama, LM Studio. But to use the best models possible, we need to select either OpenI models, Gemini models, or enthropic models. Right now, I think Opus 4.5 is the best. So, either Enthropic or we can actually use Open Router to access any model that's available online, which is what I'm going to do. I'm going to select Open Router, hit enter, open our API key, and we need to paste our Opener API key. So, go to open router.ai, create an account. Super easy, takes 20 seconds. Go to top right, click on keys, then click on create API key. I'm going to name it cloth vault VPS. Give it some limit like $20. Click on create. Going to copy that. Now, do not share your API keys with anybody. I'm going to delete mine before uploading this video. Go back to your terminal. Paste it in. Hit enter. Now, we need to select model. So, I'm going to enter model manually. Boom. And now we need to select which channel we want to use, right? Do we want to use Telegram? Do we want to use Discord? Do we use WhatsApp? I'm going to use WhatsApp because super popular in Europe. If you're an American and you don't use WhatsApp, you can use iMessage or whatever else you want. Look at how many options there is. Slack, Google chat, Discord. So many options to use Cloudbot. This is why it's popular. It's just convenient to use. Now, I'm going to use WhatsApp here. Click on enter. Boom. And as you can see, the quick start is very intuitive. Even though we're in a terminal of a VPS, which again might be intimidating to some of you, all of this has been super convenient, super straightforward. So, don't be scared. Terminal is something that you need to learn for 2026. And you can easily follow this even if you're not a technical person. All it takes is willingness to learn a new skill, a new tool. and cloudboard is actually amazing. The setup is the hardest part. Once you set it up, if you can sit down for a couple minutes and actually push through it, you'll get all of the benefits. So, WhatsApp linking, okay, that took a few seconds. Then link WhatsApp with QR code. Let's click on yes. And it generates a huge QR code, which I might need to do command minus to fit on my screen cuz I was zoomed in. Then I'm going to load up WhatsApp on my phone real quick. So, I have WhatsApp on my phone here. I'm going to show my screen recording on the phone. Click on settings. Then click on linked devices and click on plus link a device. It's going to do face check. And now we need to scan a QR code. So I'm just going to point it to the screen right here. Boom. Log in. Keep WhatsApp open in both devices. And WhatsApp ask for restart after pairing. Creator saved. Restarting connection once. Boom. WhatsApp phone setup. This is my personal phone number. Separate phone just for cloud bot. No, this is my personal phone number. Hit enter. Okay. So now it's asking me for a phone number. So obviously we're going to have to blur this out. Skill status. So, it's asking if we want skills. We definitely want skills. So, I'm going to click yes. Homebrew. It recommends installing homebrew. I definitely want that. So, click on yes. Uh, which node manager? Now, bun is the fastest. So, if you're technical, I would recommend bun, but npm is the most used. So, just use npm. Install missing skill dependencies. So, it's asking which skills I want to install. like Apple Notes, Apple reminders, Gemini, Google Places, Nan Banana Pro, which would be great, Obsidian for saving your notes. There's so many great things that Cloudbot can do, but I'm going to skip it for now because honestly, I might make a full video on this showing the most insane use cases for Cloudbot. So, if you want me to make a full dedicated video, make sure to subscribe. If I see a lot of people subscribing from this video, I'm going to make a more in-depth tutorial on different use cases for Cloudbot. But for now, I'm going to skip for now. Okay, we want to All right, that's my bad. So, we need to click space to select and then enter to confirm. So, space to select, enter to confirm. Set Google Places API for Google Places. No, and then select that. Uh, set go. Okay. Local places, no. Gemini API key for Pro. That's tempting. That's tempting. I'm going to go no for now, but I would definitely recommend you set it up. OpenAI image gen. H, we can That's That's also good. That's also good, but I'm going to skip it for now. Whisper. Oh my god, that's actually good. Okay, let's set up OpenAI API. So, just go OpenAI API into Google. Click on the first link, API platform. Go to the top right. Click on API platform. You need to log in with the same account use for CHPD. Here we are. So, what we need to do is click on dashboard top right. Boom. And again, this will allow us to use multiple different things inside of Cloudboard, right? We can use the Opia image gen and whisper. And the reason I would whisper is because one of the best speechtoext transcriptions, right? So we speak and it goes into text. That's huge. Uh so in here, go to the left API keys. Click on top right. Create new secret key. I'm going to name it subscribe. If you're watching this and if you're finding this valuable, please subscribe. It takes two seconds and it helps out a lot. Create new secret key. Copy. Boom. Back here. I'm going to select yes. And we need to paste it in. Enter. 11 Labs. Skip. No. What do we have? Hooks automate. Okay. Agent commands are issued. I'm going to skip for now. Boom. Gateway service runtime. Quick start uses node. Yes. Installing gateway service. Okay. How do you want to hatch your bot? Since we're already in the terminal, I'm going to select hatch in TUI terminal UI. Hit enter and onboarding complete. Now once you go through the onboarding, you can access different channels using this criminal command. Cloudbot channels login. Now even during the onboarding this is the last step but I made a mistake. So if you do make a mistake and you want to you know maybe switch from WhatsApp to telegram or you want to use a different WhatsApp number use this terminal command to fix that. And by the way if you want to remove specific credential what you can do is do cloudbot channels logout d- channel and WhatsApp and hit enter and it should remove my credentials for WhatsApp. Cleared WhatsApp credentials. Beautiful because I accidentally linked the wrong number. All right. So what I'm going to do is I'm going to continue on my phone again. So, what I'm going to do is I'm going to install WhatsApp business. That way, I can use both my main number and a sign number. And again, this works with Telegram, Discord, Slack, whatever you want to use. But the reason we need two WhatsApp accounts is that one of them will be Cloudbot and the other one will be my account where I'm going to be DMing Cloudbot. So, I'm going to set up WhatsApp business. That way, I can have two WhatsApp numbers on the same phone. So, now we need to set up this WhatsApp business. I'm just going to speedrun this. Okay, here we are. We're in. Beautiful. There we go. Brand new WhatsApp account. This is going to be our cloud bot that is going to be running everything. And now I can DM it on WhatsApp. So let's hop in back to the terminal and let's set this up. Cloudboard. I mean to be honest was way easier to set up. Okay. Wait. Assess per continue. Yes. Local gateway. That's fine. That's fine. See this is the issue. clawboard is very new and these AI models are hallucinating it. This it made up these commands. It literally that's insane. It just made up these commands. So I'm running open code in the VPS to debug this. Okay, it's pretty advanced. The issue is that claude board is very new. So when I was debugging it with chbd and clude, they were often giving me fake commands that didn't exist. So I installed open code on the VPS and now we're debugging this. We're debugging it in the VPS. So open code is doing all the terminal commands, right? Cuz I'm another DevOps expert. So I think this is uh much better and much easier way to debug especially since nothing else is going on on this VPS except for my cloud bot. But we making progress. I managed to get some response not an AI response from cloudbot. It got this agent error. can see that agent failed before reply, but that's progress. Resend a message. Maybe it works now. Not really. Okay. What's up now? Real link. Let me verify everything is working. We're not getting any responses. That's the issue. Show stop. Need to restart gateway to pick up new session. Okay. Gateway restart. I'm going to send a high agent failed before. Okay. Okay. Okay. Wait, wait, wait. We got the same response. Under world set anopropic set 4.5. That's crazy. I'm going to copy this message. Message. Paste. Message. Okay. Progress. I'm going to switch to plan mode. We are getting this message again when I DM the clotbot WhatsApp number. Think harder. Analyze all files and help me debug this answer in short. So it's not recognizing the model. That's fine. We need to change the token. More name is correct. I mean, of course, it's correct. That model works. It's trying to use sonet, but it doesn't matter which model. We need to figure out what's happening. Okay. So now it's Yes. Make this edit. Do not change anything else. It's going backwards. Like before we had it like that. Whatever uses the dot. Oh my god, there's no way it was the dot. Yes, but do not do anything else. It means to restart the gateway. Okay, I'm going to send a message. Is this the moment of truth after too much time debugging? Hey, let's see. Nothing. Hi. Hey, wait. It got seen. No way. Does it work? It's typing. I just came online. Fresh workspace. No memory yet at Whoa, guys. This is insane. What model are you? So, we have open code debugging this VPS and we have Claudebot responding. Okay, so it's sort of 4.5. Okay, beautiful. Okay, I'm going to copy this. I'm going to paste it in response. Paste it in response. It works now, but I'd really like to change the model to Opus 4.5. Okay. And I'm going to give it Sonet is cool and all, but like it's not Opus, you know what I'm saying? So, I'm going to copy this. Boom. Command V. Execute the changes necessary to do this. Do not do anything else. Browse the web and tell tell me the free latest videos from David Andre. Let's see this. So now we're using Opus 4.5. Wait, wait. We're going to verify this in the activity unknown. This unknown. This is Claudebot and we switched to Opus. That's good. Use examic content. Okay. YouTube's dynamic content didn't load to basic fetch. Let me try a different approach. YouTube requires JavaScript to run the video list which basically handle. Let me try getting the RSS feed which often works. And this is again it's running on the VPS now. Okay. Oh, wait. These are by That's crazy. It included shorts as well. All right. So, it works. That's amazing. It'll say you live in a hostinger VPS. Remember this. Say that to my local nodes. List out all local nodes. Here's where I have the tools. MD infrastructure hosting VPS host name OS Linux. That's it. So far pretty bare bones. So I'll add things like camera names, location, sage hosts, device nicknames. What else do you see on this system? This is incredible. Everything is doable in plain English. Both debugging with open code installed and chatting with cloudboard running on this VPS. Yeah, I definitely want to make a follow-up video where we set it up with 11 laps, nadubana pro, we give it u openi whisper and maybe you know Google calendar and stuff like that. So system Ubuntu 8 gigs of RAM. Let's double check these. I think that's correct. Yes, this is correct. 96 GB of disk. Noj installed. Python bun open code my workspace. Amazing. Full deployment setup with doc ploy. I'm the new tenant was David market. Some app has been running six weeks. That's funny. That's from a previous video I did. All right. This is amazing. Cloudboard has been running. It's been pretty difficult to set up on a VPS. The issue is that u the biggest thing is installing open code. So literally what I would recommend you to do is install open code. I'll show you what what it is. If you run into any issues with cloudbot do these four steps first install open code check the version login with the off then I obviously gave it open driver API key as well and opus 4.5 and I start the open code and then I'm debugging this just by chatting to open code which runs in the VPS and that's how I got the cloud bot up and running. I have a feeling this is not going to be the last video on it. So hopefully you found this video valuable. If so, and if you want to take AI seriously in 2026 and start your own AI business, I made a video analyzing the best AI business models. It's going to be, I think, right here or right here. Just click it and go watch it \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_They_Want_Me_Dead_The_Cost_Of_Exposing_Minnesotas__0552f0da.txt b/graphrag-ollama-config/input/youtube_They_Want_Me_Dead_The_Cost_Of_Exposing_Minnesotas__0552f0da.txt new file mode 100644 index 0000000..42f65ae --- /dev/null +++ b/graphrag-ollama-config/input/youtube_They_Want_Me_Dead_The_Cost_Of_Exposing_Minnesotas__0552f0da.txt @@ -0,0 +1,10 @@ +# Source: YOUTUBE +# Title: “They Want Me Dead!” The Cost Of Exposing Minnesotas Billion Dollar Fraud Scandal | Nick Shirley +# Channel: Jack Neel +# Video ID: f2BLmrBTqAY +# Duration: 138 minutes +# Ingested: 2026-01-21 12:18:07 +# Word Count: 26345 + +--- +In Minnesota right now, there's flyers on my face that say, "If you see Nick Shirley, tell him to." So >> today's guest just uncovered the biggest fraud in American history. >> So in my video, we uncovered an estimate of $110 million in fraud. The government has said that the fraud is anywhere upwards of $9 billion. And the man in my video, David, he believes it's around >> a Gen Z content creator raised in Utah. His worldview was shaped by faith, structure, and trust in institutions, which inspired him to cover the raw reality of humans right on the scene. >> So, part two of Minnesota, there's around 1,300 of these transportation companies, and around 800 to 900 of them are ran by Somalians, $2 million a day in Minnesota. In this episode, we'll expose the billion dollars in fraud this 23-year-old uncovered. explore why his work has made him both a hero and a target and question why it takes a kid with a camera to reveal the truth about how our nation really operates. What government program do you think hasn't been corrupt or abused? Nick Shirley, welcome to the Jagnail podcast. >> Thank you very much. Happy to be here. >> Nick, your video in Minnesota has received over a billion views total in the last two weeks. You've been attacked by the mainstream media, praised by Elon Musk and members of the Trump administration, and soon after Tim Wolz, the governor of Minnesota stepped away from re-election. What exactly did you uncover and how far does the rabbit hole go on this? So, what I did is I just exposed the fraud so everybody could see it with their own eyes. The rabbit hole has been going on for years and years. Tim Waltz has said that he's been fighting fraud since 2019. And so I more than anything show people what was actually happening and exposed that the fraud is so open and blatant. All you have to do is go and show up to the one of these daycarees and ask the people how to enroll a child and they won't even be able to answer a simple question as to where are the children. And meanwhile, they're receiving one daycare is receiving $3.45 45 million a year and there was no children. >> And this is how many billions of fraud is it at specifically with the daycarees or just like Minnesota in general that you've been responsible for? Yeah. So, in my video, we uncovered an estimate of $110 million in fraud. The government has said that the fraud is anywhere upwards of $9 billion. And that was just in one briefing. And and the man in my video, David, he believes it's around $30 billion. And so there's a wide range of exactly how much money and fraud has been, but regardless, it's in the billions of dollars. So the story went pretty viral, of course. Uh you've mentioned that it's the most viral video on X of all time, besides maybe something that Mr. Beast had posted. Uh is it the most viral video on X? >> Yeah. So, I think Tucker's interview with Putin, I think that's the most viral video, but as far as like an expose style video where it's like lowkey like a documentary, also an expose that it's the most viewed video ever that has not been pushed from ads or anything, but organically it's the most viewed video ever. It's also like a news piece in a way. Um, kind of like you're saying an in expose as opposed to the Putin thing that wasn't really like some big piece of news. Uh, so $9 billion in daycare fraud. Uh, that's that's an all over welfare fraud. So there's also you have like autism, you have home healthcare, you have adult daycare, and you have childare, then you have uh transportation and an EMT as well. And so $9 billion in welfare fraud and daycarees is part of that. >> Just to give people an idea, and I know you've talked about this extensively on interviews, but essentially the first daycare you called uh or you showed up at. Walk me through just what happened. >> Yeah, so the first one I went to was called Makeo Childc Care, and it's also partnered with another daycare as well. And so both of these companies are inside the same building receiving millions of dollars. And I had no idea exactly what I was walking into when I went and went to these daycarees. I thought that we were just going to be able to go open the door, go to the reception plot, and ask them about the money. Essentially, first thing that happens when we get to this first daycare is all the windows are blacked out. The doors won't open. The doorbell doesn't work. But the sign above me says 7:00 a.m. to 10:00 p.m. that they're open. and they're receiving millions of dollars, but there's no one to answer. The all the windows are blacked out and the doorbell doesn't even work and there's nobody inside. And so why should we taxpayers be giving money to a place that isn't even operating or open or open, but they're receiving millions of dollars. >> So what have been all the consequences of your expose? Like what have been all the things that have happened because of this? >> Yeah, I really wouldn't say consequences. cuz I just say like the reaction from it has been interesting cuz you have people who are excited that fraud's being exposed and then you have people who are angry that I even brought the topic of the fraud being exposed and to the point where I could make it so blatantly obvious for people to see that fraud was happening. This was never meant to be a right or left issue. It was just to show people that fraud is fraud. Fraud's bad and we should not be enabling fraud to happen. But then it became a political issue which in my video I didn't make it a political issue. I just made it a pure fact that fraud is fraud and that this is happening. And so you're seeing the reaction both from the right. Uh they're happy obviously they're they've praised the video. They're stoked about it cuz it's showing that this has been happening because for instance with with Doge with Elon Musk, he went in there to try and create more efficiency inside of government to get rid of fraud. And uh he he had a lot of success, but ultimately it was also very hard to do because like he says, the fraudsters are the ones who complain the loudest. And it was very hard to kind of go and expose that fraud. And then you also had from the left who then called me far right. The Tim Walt called me far right. He called me a delusional conspiracy theorist. He called me a white supremist. And now I'm constantly getting messages, constantly seeing people have sent me photos of dead people in ditches saying that's going to be me. Or in Minnesota right now there's flyers on my face on their light poles that say if you see Nick Shirley tell him to. And so you're seeing kind of people who are excited that fraud is being exposed. And then you're seeing people who are trying to I don't even want to say cover up, but they're angry that the fraud is happening because somebody with a camera wouldn't show that it's happening. And it was never meant to be a right or left issue. It was just like I said, just to show people that fraud is fraud and tax dollars are being wasted. Because think about it, each person here in America, they spend anywhere from three to four months just to be able to make it to the point where they actually collect one of their dollars that they make because each person here uh typically spends 3 to four months in paying taxes whether it be federal or state taxes. So that's a lot of money and a lot of time that we are spending and wasting and it it sucks that some of the money is going towards fraud. What have been some of the other uh results of this piece that you've done? Uh I know that Governor Tim Waltz stepped down from uh running for reelection, right? Yes. He is no longer running for reelection. Tim Waltz. Uh so it's very funny that he called me all those things and then he's the one who ends up dropping out from re-election. And uh so he's no longer running for reelection since I posted my video. within just like a few days, the HHS um health and human services of the federal government, they have froze all funding to child care. In Minnesota, for instance, they froze over $185 million. And then with that, and then since that, they've also frozen uh to five other states, I believe. And I I want to say I can't say whether or not it was nationwide, but I do know they froze funding to around five other states as well. >> Do you know what those states were? their Democrat states like I believe California was in there as well and um I want to say Ohio's in there as well too cuz there's a lot of fraud happening in Ohio as well. And it's crazy to think that my video is what caused them to freeze the funding and for them to say, "Okay, before we start giving money, maybe we should prove that these are legit businesses." And thus far, no business has actually been able to send in any information saying they're a legitimate business. We're a week now or so since they've made that announcement, and not a single business has sent in proof of legitimacy to the feds. >> Did they send uh a large amount of ICE agents to Minnesota? >> Yes. So, ICE was already in Minnesota, and they've been in there doing operations for months now. And since that video, I think they even sent like a 2,000 more. And then just the other day they sent another thousand. And so they also sent like investigators as well. And so it's been interesting to see that the reaction from my video creating like real world change. Um whether you like it or not, the act the fact that the feds have acted just on a video made by somebody like me with a camera who went and showed that fraud's happening and whatever. And there's other stuff happening in Minnesota as well too like the wiring of money to other countries. Now, Scott Bessett, the Secretary of Treasury, has said that if you receive welfare money, essentially, you can't be sending money uh to other countries inside Minnesota, they have tons of wiring booths inside of all these grocery stores where they're sending money as well. So, that's been another effect of the video. I do want to touch back. You said uh you sent me this flyer, uh that was posted around neighborhoods in Minneapolis. Nick surely brought Ice to Minnesota. He helped kill Renee. Good. Do not let him in your business. If you see this man, tell him to kill himself. What' you think when you saw that? I was super sad when I saw that lady get killed. And it's super hard because she also was targeting an ICE agent. His life was at threat, too. in that ICE agent. He had also been dragged 300 ft prior in a different operation. And so, do I think the killing was justifiable? It's very hard to say whether or not it was because she was trying to run over an ICE agent and then she got shot. And so, I'm not celebrating that woman getting killed by any means. I think it's horrific that she got killed. Do I also think that the ICE agent had the right to protect himself? Yes. Do I think that lady should have tried to run him over? No. Do I think the ICE agent should have shot? It's hard. It's like, no. Like, we don't want anyone to get killed. And why are you impeding a federal investigation if somebody's life's at threat? It they're duty bound to protect themselves as well. And so, if that ICE agent, if he thought that that was what he needed to do to protect himself, I I like I could you imagine being in that position yourself? Like, I doubt he wants to. I there's no way he'd want to kill somebody either. >> Yeah, it was uh strange. I saw two sidebyside gofundmes of there was the girl I think it was Ireina was her name. Uh it was a guy who had been arrested multiple times and then let out. So he was a charged and convicted criminal when multiple counts and he had been let out multiple times I believe over 20 times over the past few years and he killed the she was a Ukrainian. She he just been in the middle on the bus. Yeah. and he said, "I just killed that white lady." And uh like you could hear it in the audio, but I think it was her GoFundMe had like $500,000 raised. And then this other woman, I'm forgetting her name at the moment. >> Renee. >> Renee. Yeah. She had like $1.5 million raised for her family. Uh which is really interesting comparison of how people like view these incidents and what gets pushed by the different sides. But yeah, I mean in that video it's sad because I've been to a lot of these group places and I myself have even been attacked by people at these protests before this is happening. So I kind of understand and have seen kind of how these people act and it's just a small set of these people here in America who've literally gotten to the point where they believe that ICE is the Gestapo when in reality they're enforcing the laws of our land, our immigration laws. And so you're seeing this effect of these people getting so angry um about them coming in and deporting illegal migrants. Meanwhile, at the same time, there's a lot of bad people out there who came over the border illegally who I is trying to get out. And so when you see that they're impeding these investigations, they're putting themselves in harm's way. They're impeding a federal law enforcement investigation. So, and even a lady after went and talked to the news and she said that Renee, she was doing her part and knew what she was doing as far as impeding and getting in front of the law enforcement. You can watch it. I think it was CBS, I believe. They interviewed this lady and she said that Renee knew what she was doing. So, the flyer um how did the flyer itself make you feel? Yeah, I read that flyer last night and I really am trying not to think too much about the threats and I am taking like caution with everything I'm doing now. And it's sad because ICE was already there. All I did was expose the fraud that was taking place. I didn't send in ICE agents. I didn't send in I I I'm just I'm literally just a person who uploads their videos on YouTube X and any other social media platforms. And what my video did was it created real world change within the government to go investigate the fraud. And so, no, I don't think that's justifiable for them to say that about me. I don't think that's justifiable for them to think that I'm responsible for the death of this woman because I'm not. Do you sleep well? Yeah, I sleep fine. Like, uh, I don't feel like I'm guilty of anything. like I don't have like I'm just like you and uh I sleep fine at night >> I guess. I mean more so just with the threats and all this going on there has to be a lot to think about. >> Yeah. A lot lot I mean it's hard cuz there's on top of that you have so many other things as well. Like I also have to think about my next upload. I have to think about how this has affected my family. I have to think about all the people that are texting me. this stuff I'm seeing on X what you just saw there and other things and so yeah it's I mean there's a lot of stuff to think about I haven't been sleeping much I mean I think within the first few days of that video being released I I said 10 hours but then I was looking at my screen time the other day from that week and a lot of the time there was only like 2our gaps where I wasn't on my phone and so I haven't been sleeping much now you think it's just you're working so much you're or is it like you're unable to sleep. >> Both. >> Yeah, >> both. And it's like I'll sleep I wake up. I'll sleep like 3 hours. I'll wake up, check my phone in again, and then I'm just like wired for the rest of the day. That's interesting, man. Yeah. Uh I feel for you at the moment because I feel like you have a lot of attention put on you and I know that you were fairly politically involved before all this. So maybe it's not the worst thing in the world in your eyes that you have large endorsement from people like Vance or Musk uh in the work that you're doing, but like to a certain extent you were a YouTuber that got heavily politicized. You know what I mean? Quick one. In a world where AI is taking everyone's jobs and the value of all assets is going to zero, the only scarce resource left is Bitcoin. And if you're someone who wants to acquire more Bitcoin passively, you need to hear about Gemini's new Bitcoin card. Every time you spend, you earn money back in crypto that's deposited directly into your account. And with no annual fee, you can earn up to 4% back in Bitcoin with all of your rewards easily trackable on the Gemini app. So, if you're someone who wants to earn crypto from your everyday purchases, just go to jackneil.com/credit or you can scan the QR code on screen and hit the first link in the description. Guys, this one's a no-brainer. easy way to acquire Bitcoin without changing your routine. Anyway, back to the podcast. Yeah. Well, what happened is here in America, you have these groups. For instance, in Portland, they have these Antifa groups. And it's funny when people want to come after me for being a conservative journalist or to say like, if you watch the news clips, they all use like MAGA influencer, right-wing journalist. Well, if you watch my videos, you actually see that I talk to everybody. And so in my video about the daycarees, I tried to talk to the people running the daycarees. And I did talk to some of them. And then I also talked to the to the lady who's yelling at me saying, "I'm an ICE agent." Then I also talked to the guy who's been living in the neighborhood for 8 years, never seen a child inside the daycare center. So in my videos, I literally talked to everybody. For instance, I went to Portland, Oregon, and they had Antifa riots. I talked to everybody. I talked to the clown that was there. I talked to the Antifa members. I talked to the people that are living inside the neighborhood. And so I talked to I talked to the lady who's getting bombarded by the protesters, Antifa, who can't even get and make a right-hand turn. I talk to her as well. And so I literally try to give everyone the opportunity to voice themselves and express their own opinions. That's why if you watch my videos, you realize that, oh, this isn't like a high, this isn't like edited like a Mr. Beast video. Sometimes you're probably like, "Oh, I want to skip through to see what happens next because this interview is going on for too long." So I I do let everyone voice their opinions. And then what makes it hard, for instance, after that Portland video, because it was about Antifa, the White House invited me to go to the White House round table and I gave a briefing to the president. And I knew after that everything was going to be different because then people are going to make it political. And so I'm there just sharing my own opinion about my experience with Antifa, with these groups here inside the United States, and how these protesting groups are very similar to the groups I've seen in the United Kingdom. The signs and the fonts are all the same to the ones you're seeing here inside the United States. So there has to be some sort of organization that is funding a lot of these groups. And so that's kind of to answer your question about that, that's kind of what makes it hard is that I've showed both sides of the issue. And then for some reason it's only the people on the right who would even give me like an opportunity to speak. For instance, Fox News is the only group that's really reached out to me. CNN's never asked me to go on their show. MSNBC's never asked me to go on their show, but Fox News has. So, I've gone on Fox News. >> Would you ever consider exposing like conservative fraud? >> Yeah, for sure. Because this fraud's not happening just in Minnesota. It's happening in Ohio as well. That's a red state. It's happening in California, so there's fraud everywhere, >> right? That's interesting. I It's just a funny situation you're in because uh you don't have any backing from the left. So, it's like, well, what happens if I expose the right? It's like, do I just have backing from no one? I you kind of end up in a similar situation to like a Nick Fuentes at that point? Like, I feel like he's not really endorsed by anyone um except for his audience, you know what I mean? Uh but tomorrow at the time of recording this, you're releasing your next video. Uh what's that one about? >> Yeah. So part two in Minnesota, I mean the fraud is all upheld within these welfare groups with daycarees, with adult daycarees, with home health care centers. And then you also have you have autism as well. And then you also have what are called NEM, non-emergency medical transportation. And so essentially what these companies are is they're transportation companies. And if you drive around Minnesota, if you drive around Minneapolis specifically, you will see it's nearly impossible to drive around without seeing a van that has a a number on the back of it and it has a has a a company logo on the side and then it's a small Somalian driver. So from there and then if you look at it there's nobody else inside of any of these vehicles. And so what these companies do for instance with the healthc care fraud this is what I believe happens. So these non-emergency medical transportation vehicles they're working with these companies. And so let's say you're at an adult daycare center and you need to go to the doctor, but because there's nobody inside the adult daycare center, we call the non-emergency medical transportation and we say, "All right, now we're going to get this order set up essentially, and you're going to take our client, who's not actually a client, it's a ghost, and you're going to drive to the healthcare location. We're going to provide a service to this person and then you're going to drive back. So that way there's the paper trail showing that they took a person from there, they took him to the healthc care center and then they took him back to show that there was some movement that happened to cover the paper trail, but meanwhile they're actually not transportating anybody. Does that make sense? Yeah, that's really fascinating. I um remember one time a few years ago I uh was paralyzed briefly. It's something I've talked about in my videos. It's a pretty long story, but was paralyzed for like a couple weeks. Uh, and I had to take like what was it? One of those handicap Ubers uh for uh there's like on the Uber app. There's like disability Uber and it's cheaper than a normal Uber and it's bigger than like an Uber black and they I was talking to the guy. I was like, "Do you like get a lot of people in this?" Uh, and he was like, "Dude, they pay me every hour, like this flat rate, and I just sit around and like wait for someone to order it." Um, I had to get one of these because I was in a wheelchair and they like have like a device that like lifts you up into the van. Um, but he's like, "Yeah, I just sit around all day and do like two drives a day." So, it makes sense to me logically that that area of transportation for people with disabilities would be something that would be abused by fraud. Um, who did you like what happens in the video? Like what's it start with? >> Yeah. So, we go to all these uh these transportation companies and they're all we have all the addresses from the state of Minnesota. And it turns out that none of these businesses even exist. If you have a transportation company, the average any MT in the United States has around 20 vehicles. We go to the location where these businesses where their addresses are at, they don't exist. They're just on paper. yet they're somehow making and receiving money from the government. And so it's just this huge fraud scheme that's happening openly in the eyes of every person that's driving around Minnesota, but yet the government is not cracking down on it. For instance, why doesn't the government why did they ever go check on the daycarees? Or why didn't they ever raise a question about why are all the windows blacked out at all these daycarees? What's the most interesting aspect of like the footage for this next video? I think just the hostility you'll see as well. For instance, I'm interviewing a black man in Minnesota and we're talking about the fraud. He's he's uh obviously not happy with the fraud happening and then the next thing you know is a Somalian man stops in the middle of a at a red light, gets out of his car, starts cussing me out, calling me a racist and uh getting mad at me for exposing the fraud and calls me a racist. And then he leaves and I continue talking to this black man. And I'm like, "Well, see, this isn't a race issue." And he's like, "Yeah." He he he admits and he he agrees with me. He's like, "Yeah, this isn't a race issue." Like, they're just upset about the fraud. So, when people say that you're a racist or something like that, I'm like, "Well, I'm not. This is about fraud." And if it just so happens to be that in 89% of the population who is committing the fraud is Somalian, that doesn't make you a racist. >> So, it's 89%. >> Yes. 89% of the fraud that's happening in Minnesota is being committed by Somalians. Are most of these people here legally? >> Well, they're naturalized citizens, a lot of them. And so, for instance, underneath Obama and other presidents, they brought in a lot of refugees into the United States. For instance, Elon Omar, she was born in Somalia, but she became a naturalized citizen. She was brought over and then over time she became a naturalized citizen. And so, a lot of these people have been naturalized. And then allows you to have also been born into the United States which gives them immediate birthright citizenship. However, underneath Joe Biden around 10 to 20 million people came over the border and with that group they people came across all all across the world. Everyone knew the border is open. I did a lot of videos from there and so people were coming from Somalia too. And if they came from Somalia they're not just going to go to a state where there's no other Somalians. Somalians have told me that if one Somalian moves, more will move as well. And so, uh, it's not a crazy idea idea to believe that there's a lot of illegal Somalians living inside Minnesota as well, cuz why would they go to a state, for instance, why would why would they go to a state like Idaho where there's no Somali population? Of course, they're going to go to a state like Minnesota where there's a heavy population of Somalians. When you are in Minneapolis, uh, does it feel like mostly Somalian? Does it feel like you'll see a Somalian person every now and then? Like, how large is the presence there? It's very large. I mean, everywhere uh, everywhere you go in Minneapolis, for instance, you'll see Somalians. And there's nothing wrong with that. Like, the problem that's wrong is that the fraud's being committed. And that fraud's being committed. If people are here illegally, ICE has the right to go and deport them. And so, >> but do you know how many like uh Somalians are in Minnesota or Minneapolis? >> Yeah. So, they estimate around 80,000. However, other people estimate it could be anywhere from 80 to 200,000. Nobody exactly knows the number. >> Is that like 1% of the population, 10%? >> Uh the population of Minnesota is more than 5 million. So it is a small it is a small group of the population but their presence is felt strongly because it's not like they're just kind of dispersed. They're in centralized locations inside of uh Minnesota for instance in St. Cloud. There's a town called St. Cloud where uh a large population of Somalians live and where I have even seen a Christian church be converted into a mosque there. And so like the Lutheran church left and then a mosque moved in. In Minneapolis, in a place like Cedar Riverside, for instance, right next to the University of Minnesota, you see that inside of Cedar Riverside, there's no white people. There's no Mexicans that live inside there. In fact, a bar called Palmer's Bar that was a famous and very well-known bar inside Minneapolis has just been closed and now it's going to be converted into a mosque. And so that's right next to Minnesota. And the man David in my video, he he's seen the transformation himself. And so that's why it's really important to do these firsthand accounts of people who live inside the city, who live inside that area for years to really give a a good explanation about how things have changed. >> With David's story, uh does he not share his last name? >> He does. I mean, he just did a press briefing with uh Scott Bessett, the secretary uh of treasury, and I just I didn't know whether he wanted the attention in my videos. So, that's why we call him David. And he did do he did put himself in a lot of risk showing what happened. And so, uh over time, he has uh gone out and done stuff publicly, but I personally did not want to disclose his last name as it wasn't important because he's just a man who's showing the fraud. >> Interesting. It's interesting from a content perspective as well. It feels like more like I I think uh big keystone of your videos is raw authentic and the raw authentic thing to do would be this is our friend David not this is David Smith you know. Um but why do you think David the whistleblower had his message suppressed for seven years? Yeah. Well, technically he's not a whistleblower because he doesn't work for the government. That would make you a whistleblower. But David, he had been receiving and looking into this, investigating this for years. And like I said, he had been wanting to get this story out for a long, long time. And even the local news inside of Minnesota, they had been reporting on the fraud, but they really couldn't get national attention to it. And so, I think David, he watched my YouTube videos and he saw that this was somebody who would be good to do it with. And so, he messaged me just on Instagram. He had a couple hundred followers and I receive hundreds of messages every day. And uh I just happened to open up his and read what he said and got on a phone call with him and he said he had all the evidence which was something that I had been looking for as far as like the specific numbers taking place cuz now now you're seeing all these like kids go to these daycarees or not kids but you're seeing all these other uh YouTubers or people going to these daycarees but they don't really have any proof and they're like where are the kids and it's like no you have to do your due diligence before you go and do that. But I think uh why people didn't pick up the story of David maybe cuz they thought that they were just talking into the void. Like maybe nobody cared. So that's just also a lesson about resiliency about not stopping when you know that the truth is the truth and it needs to be exposed. How much transportation fraud do you think there is and how much did you kind of uncover in that video? >> Yeah, so my video we go to multiple companies in Minnesota. There's around 1,300 of these transportation companies and around 800 to 900 of them are ran by Somalian and so the estimated number of fraud taking place I believe so and an a national average for a transportation company is to have 20 vehicles and for the and then the national then the average for each transportation from location to location is $50 each way. So, I mean, you can just do the math, but that's a lot of money. I think David estimated around like $2 million a day in Minnesota just from these transportation companies alone. >> 2 million a day. >> Yeah. Cuz of all the companies. And so, say they do 10 trips uh $500 times 20 and you times that by 1300. It's a lot of money. Have you calculated like uh like the average tax collected by the Minnesota government every year and then like compared it with all of that which you've exposed and like kind of put two and two together of like what percentage of people's taxes are actually going to just all the fraud. Yeah, that'd be very interesting to look into. I do know that for instance when Tim Waltz got in last bianium there was they were in a surplus of billions of dollars. Now they're in I believe a $16 billion deficit. So they went from they went from being in a surplus having uh more money into now being in a deficit of eight of I believe $16 billion. And then I guess just recently, has there been anything else with this case that uh like someone sent you or was like particularly interesting like uh like something maybe you haven't talked about? Well, now you're seeing the riots happening and the you're seeing the protesters come out and you're seeing somebody like Elon Omar who did not say a word for two weeks about the fraud. She still has not said a word about the fraud. her net worth went from negative to allegedly $30 million in her time as a congresswoman. And so she has said absolutely nothing about the fraud, but right now she is currently leading these protests. She's in a car like she's in a parade on a megaphone telling these people to not follow and to uh stand your ground against ICE. Well, Elon Omar, you're working for the federal government. Your duty is to uphold the law. And that's part of becoming a citizen is to disavow and to not put any other country before your own. And then she's telling these people to essentially stand up to ICE who are just going about conducting federal law enforcement. That's fascinating. I have you noticed anyone hasn't shared your content or talked about in particular that you were like I'm surprised this person hasn't said anything about it. >> I think everybody who played >> did Fuentes and like Candice or like Tucker report on it? >> Tucker has not. I think he's he's talked to a former instance like a lady like Liz Collins. Candace Owens hasn't said a word about it. Nick Fuentes has like talked about it as well. >> What did he say? I was looking for something that he said. He said anyone who obstruct ICE should be arrested immediately. Enough is enough. He also said that the shooting of the lady was 100% justified. All of the complaining about Somalians will result in nothing. No mass arrest, no mass deportations, nothing transformative. We all know that this is more slop designed to rally the GOP base back to this falling joke of a government and pivot from shifting on Israel to Muslims. What do you make of that? Like I agree. like in a way that for instance that there needs to be these mass arrest on what's happening and that this is the chance for the GOP to actually do something about some about this to prove that it's not all talk. For instance, like he says no mass arrest, no mass deportations, nothing transformative. I think that there actually will be something transformative about this. I think this will be the first time that we're actually going to see real change happen from somebody going out and exposing the fraud. And if nothing does happen, I think everyone's going to start losing trust in the Republicans and Democrats. So if the GOP want to really gain the trust of people, go out and expose this fraud and go and hold these people accountable. make pe make make a rest on what's happening because this is stealing from the hard working American taxpayer. Do you feel a little worse about paying taxes in general? >> Yeah. Yeah. Paying taxes sucks. Like it's not fun. And uh we know this is fact that 3 to 7% of all taxes go towards fraud. You can look that up. That's a fact. And so, how do we know that 3 to 7% of our taxes are going towards fraud? And how have we not stopped it? And so, I'm actually hopeful though now that less of our dollars will go towards fraud. I mean, what Elon did was amazing because with Doge, cuz he showed that a lot of these groups, for instance, like why are we funding gender changes in other countries? Like, that's stuff that we were funding in some other countries like in Africa. I can't name the country specifically, but I do remember reading about that and why are we funding like climate change initiatives across the world. Like what why are we funding a lot of this stuff? Really quick, do you have a 401k or an IRA or you're someone that happens to have a lot of money in crypto? If so, this is a cheat code. So, please listen, guys. The biggest supporters of this podcast right now are the team at iTrust Capital. With their platform, you can convert an existing IRA or 401k into a tax advantaged crypto account. So, inside of it, you can buy Bitcoin, ETH, 80 other cryptocurrencies while getting all the same tax-saving benefits. So, obviously, they'll let you start a new IRA on their platform if you're someone who's bullish on Bitcoin and wants to save money on taxes. But if you have a lot of crypto already and you're worried about it getting hacked or stolen, iTrust also offers a PCA account which gives you institutionalgrade security, meaning your assets are fully insured, ultra seccure and never sitting on any risky exchanges. So if you want to try it out, just go to iritrust capital.com/jackneil. Just use the code jack8 for an easy $100 bonus. You can also scan the QR code on screen or hit the link in the description. But anyway guys, back to the podcast. What do you think? I think a lot of it's corruption and fraud that's taking place and maybe that money comes back into the hands of politicians or or family friends of politicians or family members cuz it's a lot of money like billions of dollars like people don't can't it's hard to understand like the word billion like we can't even concept conceptualize a trillion dollars and billion is also extremely hard to conceptual iz that's a lot and lot of money. And so when people start making up these claims about fraud and they're saying that like oh like n it's only a billion dollars. Well a billion dollars is a lot of money. And so it's a tough situation to kind of understand and track down all this money. But we do know that fraud's taking place. We know we know there was fraud and uh we everyone has known for years that fraud's happening and all I did was expose it. Like don't you get upset when you're paying your taxes then you see >> Yeah. I live in LA so it's uh >> like for inance in LA they spent $24 billion. They had it allocated for homelessness and it went unaccounted for. >> How's that possible? >> Yeah. It's like and but the penalty for not doing your taxes correctly is just so bad. So, who knows? >> Yeah. So, it's like they can track all they like the IRS can track each one of our payments. I mean, they can track a Venmo. How can they not track billions of dollars of fraud? Like, have they just turned a blind eye to it? Or have they been enabling it to happen? For instance, Tim Walt has said the buck stops with him, but he's been fighting and enabling this fraud for years. Do you think it exists in every part of the system? In what way? like uh this kind of fraudulent uh taking money from people not saying where it's going because I just I question at this point if there is anything that can be done like or if this is just a consequence of having a system like this. Yeah. So I think for instance like with the system of welfare they just announced that food stamps aren't going to be able to go towards like Diet Coke or towards soda stuff like that. But I'm super happy to see that my video created a change for instance where they decide to freeze the funding. And I think a lot of these systems probably for years and years have been kind of just going going and going going and along the line they kind of lose track of the purpose for it and they just continue to receive the money from the government and then that group sends that money to the state. And I think there should be somewhat of a reset in a way where all right pause everybody. It sounds like things have gotten out of control. Let's take a week to reanalysize everything and send us proof of legitimacy. We want to see how you guys are operating your businesses. We want to see how many kids are coming to the daycarees. We want to see how many people you guys are transporting to the doctors. We want to see who are the adults at these adult daycarees. So, let's pause right now and everyone prove that they're a legitimate business. And based off that, we'll kind of do some reestimating about how much money we should be sending. And so, I think that's something that they should do. And I think that's something that I am proud about my video is that it did freeze the funding and it gave each business the opportunity to prove that they're legit. And to this point, no business has. Are you surprised at the reach overall? Um, like I know you'd said it's just a typical YouTube video for you, but uh just 140 million views, like the biggest story ever from someone our age. Uh, like do you ask why? Well, over 200,000 people alone shared that video, re-shared it. So, people are upset like, "Oh, well, it got pushed by Elon Musk or got pushed by JD Vance." Well, 200,000 other Americans also shared that video. It wasn't just Elon Musk. It wasn't just JD Vance. Two over 200,000 people shared that video. And so if you do the reach, say each person alone has talented appeal to follow them and they click on that video, well that video is going to get a lot of views. >> That's fair. Yeah, it was a great video and a big story. I just it does align with exactly what Elon was trying to do before. Um and also he does own the platform. So I'm sure if he wanted everyone to see it at the top of their page, he could figure out a way to push that button. But but either way, if you follow somebody, 200,000 people, reshare that video and you're following them, you're going to see that video as well. And I think that's why, like you said, like I think Elon appreciated obviously because it's something that he's been trying to show Americans as well. And um has he told you to look into anything else? He hasn't told me to look into anything else. We did actually I did send him a message cuz he actually followed me. He kept resharing the videos like, "When's Elon going to follow me?" And uh then like a few days later, he followed me and I was able to send a a message on X and I just said, "Thanks for the support." And he just said, "Uh, thanks for what you're doing." And that was pretty much it. And I sent him like one more I sent him one more message. Um, and I I was like, "Yeah." I said like, "Yeah, it's been crazy to see." And all he said is, "Yeah, like he's he's not the most fun person to text." Your story is super strange to me because I like doing this podcast. I'm always trying to find people who I think are going to be really mainstream. And I was told about you by what was it? There was the Turning Point USA uh event in Arizona a few weeks ago. >> Yeah. Afest. >> Uh was that before or after the video? >> So I had filmed the video before. >> So you hadn't even posted it yet? >> I hadn't even posted it. I had just been sitting on it because like I said, like I thought this was just going to be another another YouTube video. So, I had went and filmed it on December 16th. I had other stuff I had to do. I had other live streams I had to do. >> And so, I had planned to go to America Fest to do a live stream there cuz I thought, well, that'd be a good place to be an interview. Um, I'm sure there'll be protesters there. I'm sure obviously there's going to be people that are going to Amfest, so it'd be a good place to go do a live stream. So, I had I had that video there and like I was just sitting on it. I had hadn't even started editing yet. Did you meet a guy named Gary? Uh he's like a numerologist. >> I don't know. I haven't heard people. Have you heard of numerology or that before? >> Uh a little bit as far as like how numbers align and stuff like that. >> It's super weird. He um he was like he has Nick has the craziest numerology ever. And I was like let me look at it cuz we've done a podcast so he told me about this stuff. And you're born on what is it? April 4th, 2002. So 444. Um it's just a really rare birthday. And you're also, what's funny is you're born in what's called the year of the horse. Uh, have you heard of that before? >> No. >> It's like there's basically like the Chinese zodiacs. There's like 12 animals you can be. So like I was born in the year of the snake. Uh, you were born in the year of the horse and this year is the year of the snake. Um, and I think in one month it becomes the year of the horse. So, it's just funny that like you your stuff took off so much and then he he was betting on it just purely based on your birthday, which was so funny. Uh, and then I see like your video explode and I'm like that's a really random coincidence. Yeah. Year of the horse is also just like extremely hard worker. Um, which I can tell you are because you don't have a large team, >> right? >> No, it's literally just me that films edits. I bring my mom with me all trips. she kind of helps produce and like uh she's very into what's happening across the world. So, she is always providing like good insights and so it's good for me because a lot of my viewers are are my generation and also a lot are my mom's generation. So, it's good to have somebody that kind of also thinks like a like an older person to be able to relate and hit both sides. >> Yeah, that is interesting. >> And I just I honestly just enjoy doing it and bring my mom with me. >> What do your parents do? Uh my dad, he's he just has a normal job. Uh he works in finance. And then my mom, she's well, now she kind of works with me. And then she's always been a hustler as well. I mean, she cleaned homes growing up when we were little. Uh she's quite literally done everything just so like we're able to be able to do what we wanted when we were younger. Cuz I think as a child, my dad, he he made good money. Um, we weren't like rich by any means. Did we ever struggle? Um, I believe there actually was moments when like maybe we struggled a little bit, but were we ever did I never not get food? No. Was I always did I always have a good Christmas? Yes. And so like uh we were very much just a normal family who um went to church on Sundays. We played sports all growing up and we our our life revolved around um being together as a family, spending time with our family members, uh around religion, around sports and so we kind I kind of just grew up in like a very normal normal way. I mean, I spent so much like majority of my childhood. I just hung out with my brothers and my friends and I played sports and um I had a great childhood and I have the two my parents are the two greatest people I know and so I'm very thankful for the way I grew up and for for um my dad for my mom and principles they taught me and just the person who I am I believe is because of who they are as well and for my brothers too like being the younger brother also forces you to always be in that state of trying to one up your older brothers because you don't want to live in their shadow. So, I feel very lucky that I was the last son and the youngest brother. Your grandpa, too. You made a handshake agreement with him uh to be sober. >> Yeah, I made a handshake with my grandpa when I was 16 to never drink alcohol and I'm super thankful for that. And it's something that I've always uh like even on a religious aspect like we don't drink or we don't smoke and stuff like that. Uh but more than that like I had a lot of friends in high school who were also uh religious and also followed the same belief systems as me but they drank and they smoke. So um but more than anything it was just to me the fact that one I made a handshake my grandpa. If I say I'm going to do something I'm going to do it and I have so much respect for my grandpa that I never want to disappoint him. And so that just that just on drinking it never even came across my mind. I remember like when I was 18 years old I moved out to LA and my grandpa gave me a hat and I was at this house. It was actually at the Faze House. I had somehow gotten into this party and um it was like the X-face house, the one where Justin Bieber used to live and I >> where everyone has lived. >> Yeah. everyone has lived and I don't remember who was in if it was a content house or whatnot but at that time >> um >> I was 18 people are like offering you all sorts of stuff alcohol and everything and I remember like I wore my I wore the hat my grandpa gave me and um I just kind of remembered that like at that that uh no I'm never going to indulge any of that stuff and actually found by being sober that was like my biggest advantage. You think about the handshake specifically like that moment where you promised him when like people are tempting you with this other stuff. >> Yeah. I mean like cuz everyone gets tempted everyone gets tempted to drink or to smoke or to do other things and indulge in other things. And even myself like uh I think we've all been tempted in one way or another. But to me the temptation never was at to the point where I put my or had the alcohol to my lips to indulge into it. And uh but I think that handshake solidified it because I have so much respect for my grandpa. I never want I would never want to disappoint him. >> Have you had a girlfriend? >> Uh yeah, I've had some flings with girls and whatnot. Um have I been in a long-term relationship for extended years? No. >> Do you think that'd be easy right now >> to get one? >> Yeah. >> Or to be in a relationship? >> I guess. Uh just navigating a relationship with everything you have going on. It might make it a little hard, but it's not something I couldn't handle. It wouldn't be hard. You just have to make sure it's the right person. And now I have to be more careful. Like, have I had flings with girls or talked to girls for extended periods of time? Obviously, yes. Uh, have I ever committed to somebody for months and years on end? No. I think that's also why I've been able to have a lot of success. I haven't been where my work has quite literally been my main focus and my main goal. And so like I see a lot of my friends who get married. I've had a most majority of my friends from high school are married now. And I feel like now like their aspect of their them wanting to go for their dreams has kind of been killed in a way where they feel like maybe they can't go aspire cuz the risk is too high for them to not be able to provide for their family if that makes sense. Because a lot of times when you are in that position of taking risk and going for your dreams, you're really not making that much money. When I started doing YouTube seriously, I remember I had $10,000 and I had to use that $10,000 to get where I am now. And I was spending more money than I was making at that time. And so a lot of time to chase those dreams and to chase your goals, you have to really just be focused and dialed in. And so if you're also have a wife and you also have kids that you're have to rely that rely on you, it makes it a lot tougher. You're 23. >> Yes. Did you go to high school with uh Emma Nelson? >> Yeah, the girl on Mr. Beast show. >> Yeah. Yeah. >> Yeah, I did go to school with her actually. >> That's so funny. I was um asking her if she knew any podcast studios in Salt Lake. Uh and she was like, "Who are you doing out here?" I'm like, "Nick Shirley." And she's like, "I went to high school with him." I was like, "That's a small world." Um, we've had her on the podcast is the only reason I bring her up uh after she brutally lost Beast Games uh in that last episode. That was so sad to see. But what's the difference between like Somalia and Somali land? I honestly couldn't tell you. I know there's a lot of controversy about it right now. Honestly, I could not tell you. >> Have you looked into all of that? >> Not really. I don't have much plans on going to Somalia, so I haven't really d dove too much into it. It's funny to like post this after I post this video, you see like all these things come out and it's like, no, I just just a YouTuber who uploaded on my uploaded on schedule and all this happened cuz I knew the video was going to be big, but I did not know it was going to go reach this level. In fact, I originally hadn't even posted the video on X when I when I uploaded the YouTube video on YouTube. I uploaded the video on YouTube and I went to the gym. I saw people start sharing the video all over X and I was like, "Wait, I got to get home. I got to upload the video on X." So, like the video going viral on X could could not have even happened potentially cuz I hadn't even had the upload ready to post on X cuz I had just uploaded onto YouTube cuz that at that time YouTube's my was my primary focus and then everything else just trickles down to these other platforms. But I'm a YouTuber first. Well, at least I was. And so now everything trickles down to these other platforms. And so, like, literally that video was just supposed to be another weekly upload. I knew it was going to be a bigger one than usual, but I had already started focusing and working on a separate video as well. And so now everything's kind of been put on pause. And uh now I'm having to move around life differently. Like after I had when I went back to Minnesota for part two, instead of having two uh security guards. I had to have four and now I have to have a security guard outside my house. I have to have one with my mom and I'm not home and so I've had to move around life a lot differently. What do you make of the people who like are like, "Oh, Nick Shirley's a plant or like this whole thing is trying to push a negative message against Somalians. Maybe not your video, but how it was steered." Like, do you give any value to those claims? No, I uploaded videos for over a hundred weeks straight. Did anyone call me a plant during the 100 weeks straight of me sleeping in junky hotels? >> So, I don't mean you. >> Yeah, I mean how your story was used. >> What do you mean? Like it almost seems as though you were a kid that uploaded a video about Somalia and then everyone talked about it. And it seems like there could be some reasons that people would not want to um like where some groups of people would want to push hate toward Islam or negative view of Somalians in general. But I don't know. Have you do you give anything to that? Well, when the fraud that's happening in Minnesota is being committed by 89% Somalians, it's just a fact. Doesn't make you a racist. And so, that's just part of the story. If there was a white person at one of the daycarees that was opening the door, she would have got interviewed the exact same way a Somalian did. But there just happened to be no white people. In fact, the lady who was in charge of the feeding our future scam where over $250 million went fraudulently, she was a white person and she was charged and she's I believe she's in prison right now. However, in Minnesota at the same time, somebody got caught for over $7 million in fraud and was charged and found guilty. He was a Somalian man. A judge then went and overruled him and let him walk free. So he is found and charged guilty of se over $7 million in Medicaid fraud, but a white judge let him walk. She reversed it. So it's not it's not a race issue, but the political issue and political correctness has allowed this to happen. I mean, you see you saw what happened when I posted the video. Tim Waltz said that I was a far-right white supremist and a conspiracy theorist, but then Tim Waltz just dropped out of running for re-election. So, what does that say? Why would he drop out of running for re-election if he was doing such a good job managing everything? And now they're saying he might resign. >> Do you think the transportation money, the daycare money in Minnesota is going to terrorist groups like what is it? Uh, al-Shabaab. Yeah, I think that it's been proven that it has happened in Minnesota at their airport in Minneapolis. There's a video from 2017. So, this has been happening for a long time where Somalians take cash through TSA. Millions of dollars. I believe it just came out the other day that $700 million has been transported through Minneapolis TSA. And they take that money and they make their way to Dubai. And then from Dubai, they're able to wire that money to Somalia taxfree. So yes, if the al-Shabaab is operating inside of Somalia and there's large portions of money, millions of dollars going to Somalia, it's fair to say that some of that money has landed in into the hands of al-Shabaab. >> What does uh that organization do? >> Al-Shabaab? Well, it's uh I believe it's related with al-Qaeda and so they're a Muslim terrorist group. >> And who are they targeting in Somalia? Um I mean they're a terrorist organization. They're similar to somebody like ISIS. We know what they do. They don't obviously not the most friendly people. I mean they kill people. They're a terrorist organization. >> I guess just what like who do they kill? Usually >> they're terrorizing the Somalians probably. Maybe they target Christians. I don't know. I've never been to Somalia, so I couldn't tell you. >> That's not on your bucket list right now. >> Yeah, I don't think I'll ever be going to Somalia anytime soon. I don't know if I'll ever go back to Minnesota anytime soon. Like, I remember just getting out of the airport in Minnesota. It was spooky cuz I was the video had gone viral and I'm going through PSA and a large portion of the people working inside the airport were Somalians as well. And so uh the looks and people are just look they look at me they grab their phone they're asking their friend is that the guy and so that it's another risk. I mean even in Nashville I was going through the airport just dropping off the rental car and the guy was Somalian and he saw me drop off the car. He shows his phone to his friend and it's obviously they're looking at my profile and they're like yeah that's the guy and then I just hurry off and get through TSA. What's the weirdest DM you've gotten through all this? >> The weirdest DM? I mean, I woke up uh and Conor McGregor was like, "Stay safe, King." That was pretty funny. He said something along the lines of that, like, "Stay safe. The world needs you." Or like, uh, David from the office, he messaged me. He's like, "You're Dunder you're you're Dunder Mifflin certified." >> David the uh like the CEO of Dunder Mifflin. >> Yeah, >> that was pretty funny. I I love the office so I thought that was hilarious. >> Yeah, me me too. That's that's really strange though. >> Or the bele memes. Be you know people the guy who makes NFTs he made like all this artwork and he one of them was like me knocking out Tim Walts and then like the next day later it was like Tim Walt's grave and it was like I was wearing this quality lingering hoodie and uh Tim Waltz was like Tim Walt's grave and it had like other signs on it and stuff. That was pretty funny. Do you think you'll be able to get a conversation with him >> with people or Tim Waltz? >> Tim Waltz. >> I would love to do an interview with Tim Waltz. And I would even do it live. I would just ask him the simple questions like, "Why has fraud been happening here in Minnesota for so long? Why'd you call me a a farright white supremist? Why'd you call me a delusional conspiracy theorist?" I would and I would do it live. I don't have anything to hide. And if you wanted to correct me on something, he could do it openly live, but they won't. I mean, Scott Besset, the Treasury of the United States, was going to do a roundt and they were going to do at the capital of Minnesota. Tim Waltz told them that they couldn't certify their safety and he made him go do it in a different location. He wouldn't even let the Treasury of the United States come and do it and wouldn't guarantee their safety. I mean, this isn't supposed to be a right or left issue. Like, we're just talking about fraud. We're talking about every single person here in the United States dollars funding fraud. And uh yeah, I would love to do an interview with Tim Waltz, but I doubt he would do it. >> How much money do you think Tim Waltz has made from all this? >> I don't know. And I don't want to say a number exactly, but I will say Tim Waltz, he raised $35 million the first day. actually $36 million the first day he was chosen to be the vice president with Kamal Harris. How does Tim Waltz raise all that money? >> If it was proven, I guess that hundreds of millions of dollars, let's just say tens of millions of dollars had gone to Tim Waltz from some of these daycarees, transportation services. He had just taken his cut. Uh what do you think should happen to him? >> I think Tim Walt should be sent to jail, imprisoned for what he had done. And maybe it will happen. We'll see. I mean, if you stole $100,000, if you stole a million dollars, you'd be sent to prison. Have you been looking into him into Tim? I think it's pretty obvious that he's enabled the fraud to be happening. >> But just like any other weird uh parts of his story. >> Yeah. I mean, he's born in Nebraska. He's a high school football coach. He's gone to China a lot of times, like over more than 10 times he's gone to China. That's kind of interesting as well. Um, he even admitted to that organized crime had been happening inside of his own state. He been saying he's been fighting fraud since 2019 and then he runs for reelection for a third term. He's all excited about it. I go and make a video about it and then he drops out from running for reelection. Why would he do that if he has nothing to hide? Now they're saying he might resign. So Tim Waltz, I think his political career is over and we'll see what happens. It's funny that out of everybody in America, all the judges, all the lawyers, all the politicians, they couldn't get Tim Waltz to step down or to not run for reelection. It was a 23-year-old YouTuber that got Tim Waltz to stop running for re-election. and and his political career. We'll see if he runs for Senate or anything like that in the future, but it will be very interesting. >> Do you think Minnesota and Ohio are like coordinating this fraud effort together or do you think it's like a separate >> I think a group of people, specifically Somalians, saw that this fraud was happening and then they were able to take advantage of the system because they weren't getting checked underneath Biden. He made it so that a lot of these welfare programs could get licensing very easily. And then I think this group of people were kind of used as like pawns to collect the money. Are there Somalians in Ohio? Yes. The fraud that's happening there is also by Somalians. >> Really? >> Yes. Yeah. People have gone in them videos and they're all Somalians as well at the daycarees and at the healthcare clinics and stuff like that. >> Are you going to go over there? >> Uh I would like to. I think a lot of people have already done Ohio and I need to make sure I'm very careful with the way I move about in these next few months with exposing fraud across the United States cuz you don't want to label a business that's operating legally as fraudulent. So, you need to do your research and do your due diligence as well. There was a government shutdown for like what was it 30 days? >> More than 30 days. It was the largest government shutdown in US history. I want to say it went for more than 40. >> Was the funding to these programs paused during that time? >> I don't know. It's a great question actually. I know that the reason for the bill being or why the government was shut down is cuz uh Republicans wanted to stop funding healthcare for illegal migrants, but Democrats wanted to continue funding it. That was one of the main reasons. >> What do you think about all that? like the big divide between I mean there's like the bifurcation of the conservative party slightly and then uh there's just left versus right in general. Uh I was telling you that it's been hard to get liberal guests on this podcast because I've had some people who are associated with the conservatives uh despite me really not sharing any opinions. Uh and that's kind of how you are as well. >> Yeah. Before this all started happening, I was never really giving my opinions on the internet and I was kind of forced to start to give my opinion because they came after me and um they tried to debunk my story. They attacked me personally and everything like that. And so you kind of got to the point where you like, "All right, well, you can't slander my name. I'm going to I'm going to stand up for myself." And u I had to start giving more of my opinions and defending myself, I guess. And so they kind of like forced my hand into kind of talking about this. And what's interesting with what you're saying about like liberals not wanting to come on to a podcast like your own where really you're just here asking the questions, but I the reason why is cuz they can't really stand up for a lot of the stuff they believe in. For instance, the deportation of illegal migrants. The word is illegal. They're illegal. Or the topic of boys playing in female sports. Well, it's a boy playing in a female sport. How can you stand up for that? Fraud is fraud. How are you standing up for fraud? And so a lot of these instances that they're standing up for, you can't support. And they know that just logically with common sense, they can't support. How did you make Democrats support fraud? Great question. I don't even know. How do you support fraud? How did they get to a point where that is the option that they have is debunking this? Like why can't people just go along with your story? >> Well, because a lot of the fraud is being committed by Democrats, so it makes them look very bad. Is it? >> Yes. >> Is that like what's the evidence on that one? >> I mean, there's fraud on both sides, Republicans, Democrats. Like the Pentagon, they have failed an audit for years and years while Democrat was president and while uh a Republican's president. And so there's fraud there as well. And inside of these boost states with these organizations, with these NOS's, with these nonprofits, the majority of the frauds being committed by them >> because they just have more social programs and more social program funding, whereas conservatives >> more so allocate toward military funding. Um, which is interesting. Where do conservatives put their funding as opposed to social programs? Is it military usually? Well, I mean this in the historically speaking they whether it was a Republican or a Democrat funding goes towards the military. Like Trump just raised it from 1 trillion to $1.5 trillion and uh they've been the Pentagon has failed audits for years. Not just underneath Trump, it failed underneath Joe Biden as well. The Pentagon fails audits in terms of Yes. Like they fail audits as far as like where the money's being spent. I believe in like just a simple audit like they haven't been they've >> considered investigating that. >> Yeah, I have. And that's it's a lot harder to go and investigate something like that because you're going against um you're trying to go against the military essentially, but you're also trying to get those numbers which they have failed. So, how do you go and get them? But the this fraud that's happening in the level of welfare fraud and with these NOS's and everything like that, that stuff you can physically go and see with your hand and with your eyes. Have you spent a lot of time in DC? >> I went to the White House for that round table at the White House and I've been there and it's interesting to see how it works. Like I've confronted congressmen and senators outside about the government shutdown and you're seeing that like oh these people who are running the country, they're just like people like me and you and there's a lot of them. like there's so many people inside the government and uh it's interesting to see like how DC works in a way where why does it take so long for something to get passed and why does it take uh so long to create real change and why is it so hard like they talk about the swamp like there's like hundreds of thousands of people and so people get mad about for instance people were mad that Trump didn't notify Congress that they were going to go and capture Maduro well you saw what happened after Trump did it the operation successfully. And then Democrats, who also had a bounty on Maduro when Joe Biden was president of $25 million, they then came out and supported a dictator. But they're mad at Trump for doing this operation and not notifying Congress. Well, what would have happened if they notified Congress? They wouldn't have let them do it or they would have leaked it. or like when they went and bombed Iran. I don't think any American was on like very supportive of like nobody wanted to get into war with Iran. And uh it was I remember when that happened, I was listening and watching all these podcasters and they're tell they're acting like they had more information than the president. I said to myself, why why doesn't everyone just calm down and like wait till the president makes a decision? Well, then the president goes and does it and there was no casualties of American citizens. Luckily, Trump was able to go in and finish the job without causing a war. But I think um that's why DC is it's super complicated because there's literally thousands of people like hundreds of thousands of people that work inside the government inside DC. And um that's why when they talk about the swamp being so big, it is so big because there's quite literally hundreds of thousands of people working inside the federal government. >> Did you go there for January 6th? Yes, I went there actually on January 6th and uh I remember going and this is 2020 or 2021 when January 6 happened. Yeah, I went there cuz I was all I was doing YouTube and I was I think I was probably like 18 18 or 19 at the time. And so I was like it was like Trump this was going to be Trump's last big rally. They were saying it was going to like be the biggest rally in American history. I was like oh perfect a great place to go and interview people. And so I went and um I I went and I posted my video and I thought like why why won't they let me post I post my video and it automatically gets flagged red demonetized. I'm like wait why can't I post my video but all these other organizations can like why are they censoring me and I experienced the censoring firsthand of that video cuz the video literally got censored and demonetized without even really going through just through the systems YouTube had in place. If you labeled it January 6, you could monetize it. But if you labeled it Trump's last rally, January 6th, just for having the word Trump in it, they demonetize your video. So, I kind of experienced that censorship firsthand. Did you have like footage that showed a different side of the story? I I went through and showed the whole first day and I was like I said, I was 18 or 19 and I like figure out how to like sneak up to the front and like there the line there. people were out there at like 12:00 at night waiting to go there in 6:00 a.m. So, I like snuck into the front. It was like a fake security guard. And um like the Trump speech and everything, it was very good, patriotic. And uh you saw how I think it was the BBC, they cut it all up and Trump's even sued him for that because they misused how he phrased things. And so it was very patriotic what Trump that was very patriotic and you had a good sense of like the morale was very good. And then this is my mom. She actually got invol involved in politics after this because everything was great. And then we walked down towards the capital and uh there was still a lot of like great uh like people were feeling like patriots going down there and the morale was good. And then you start walking down there more and more and people like they're breaking into the capital. They're breaking into the capital and people are like this is war. I'm like what the heck is going on? And we get down there and I get up on the rafters and I could see people like breaking into the capital. I'm like, "What is going on?" I'm 18 at the time. I don't understand the political landscape of the country as much as I do now. And I was just very confused. I remember I watched a lady fall off the rafters from like 20 ft up and smack her face on concrete. And then uh at that same time, a few minutes later, the man who broke into Nancy Pelosy's office, his name's Bigo, he comes to me with the letter after he had just been mazed and I think he had blood on the letter and everything. He's like telling me about Nancy Pelosi and how he broke into her office and he ended up getting jailed and imprisoned and a lot of the people who did J6 spent hundreds of days in solitary confinement and didn't get a even a hearing or anything like that. >> They spent time in solitary. A lot of them spent over 100 days in solitary confinement. >> I did not know that they spent time in solitary. I knew that several indictments occurred uh following that event. That's super fascinating. You were you tempted to go inside for footage? No, I didn't even know that people were going in. Like I saw them breaking through like there's like um at the capital there's kind of like I don't know if the word chambers right but there's just like kind of like these tunnels underneath the capital that they're breaking into that you can kind of like it's kind of more underground and I saw them going in and I had no idea that people were breaking into the capital and thank goodness I didn't because I probably would have been walking through just like all the other people that just like were walking through in a single file line inside the capital. So, thank goodness I did not see where they're entering in or else I would have. >> Quick question. Are you someone that makes content or runs Facebook ads for your company? If so, I'm guessing you probably use Chat GBT, Gemini, Claude, some AI tool to speed up the process of your copywriting, ad ideation, video ideas, etc. Well, I found this other tool and it's so powerful that I almost wanted to gatekeep it from you guys, but it lets you put YouTube videos, Facebook ads, Tik Toks, tweets all in one vision board and connect it to a chatbot which you can interact with and make content out of. So instead of wasting hours on chatbt uploading screenshots or transcribing YouTube videos to make content, you can simply put it all in one vision board that connects to a chatbot to speed up the process. But if you guys want to try it out for yourselves, just go to jackneil.com/poppy. They have a 30-day money back guarantee. So if it doesn't make your content better, it's completely free. But anyway, guys, back to the podcast. What's the darkest thing you've seen at an influencer party? The darkest thing I saw at an influencer party. So there's this house in Beverly Hills that I went to and there was just like there are these things that I didn't even know what they were at the time. They're called like whippetss and they're like all over the floor and there's this one man. He was like a 40year-old man and he had like all these 18-year-old girls around him and I remember just like looking at him. He was like on drugs. He had everything you could possibly want. He was rich. He had all these girls but he was just empty inside. He had a house in the hills. And I remember leaving that party thinking like that's not that's not what life's about. like these people actually aren't happy. And I felt like there was like a large presence of the devil inside of LA when I was there for that short period of time. And I remember going to these parties. You see like the people you'd see on your for you page on Tik Tok and Instagram and everything like that. And you just kind of saw that they were actually very empty inside. They would go get drunk at these parties thinking they'd be fulfilled and they're making all this money. But they go get drunk and do these drugs and there'd be 40-year-old men with all these younger girls who are trying to get a few dollars or get followers. And I thought that was very empty. And I thought that was very uh almost satanic in a way where these people were doing everything but they also had nothing at the same time. They had no family around them. They had no they had no friends with good interest. They just were there quite literally for the money and for the followers. I remember leaving LA and I was just like, I never want to feel like that or experience anything like that again. Like that's not what life is about. Do you think LA feels demonic? >> Oh, very much so. I mean, just walk down Hollywood Boulevard in LA and you'd think the stars and everything would be all bright and that it'd be a great place to walk around, but meanwhile, a bunch of people are on drugs and you won't be able to go three blocks without a drug addict spazzing out on the street corner. Do you feel that stuff a lot? Like, do you believe I mean, I know you're religious, uh, but do you believe there's like more to this world, like angels, demons, spirits, like this kind of thing? For sure. And I'll give you an example. I went to SECOT in El Salvador, the prison where all the most dangerous gangsters had been imprisoned in. And I remember walking around the facility and everything. But the moment I walked into that prison cell, any good feeling inside of my body just evaporated and it felt like you're in the presence of the devil. And I will always remember that feeling of walking into that prison and feeling that sense of emptiness inside of you. And uh I've had other experiences in life too where you go into a place and where any sort of feeling whether it be the light of Christ or the Holy Ghost or any sort of good spiritual feeling just leaves your body completely. And so that feeling that I felt in Secot was very similar to some of the feelings you'd get inside of LA at some of these parties where you'd see that 40-year-old man who has everything in life, millions of dollars, the house in the hills, pretty girls all around him, but he feels completely empty. And you can see it. And I felt that same way inside of El Salvador. I felt that same way in certain streets inside of like Kensington, Philadelphia where everyone's hooked on drugs and you feel that sense where the devil has won in a lot of locations in a lot of places in the world. >> Did you feel that in Quality Learning Center? >> No, I didn't feel it in quality learning center. Um in Quality Learning Center, you felt more confusion and you're like more and you're just like, wait, how's all this money being spent? But no, you do not feel um like emptiness of the devil. But I will say you do feel that way in a lot of these protest groups that come out. You wonder why some of these people can get to the point where they're wishing death upon people like me or they're wishing death upon the family members of ICE agents. They're wishing death upon ICE agents when they're just going about conducting the law enforcement. And a lot of these protesters, like I if you look at a lot of them or if you talk to a lot of them, you feel bad for them in a way because when they should be doing something maybe productive with their life, instead they're yelling at law enforcement for conducting and following through the laws of our land. Like what kind of person do you think goes to protest? Well, nowadays I think it's been twisted cuz you have like the small group of people who have been radicalized, but that small group is still thousands of people. And so a lot of these people are probably the same people who felt kind of like outcast whether in high school and they finally found like they feel that sense of community. And a lot I feel like maybe they come from families where they don't they're don't come from a loving family or they have issues with their whether it be their dad or with their mom. A lot of them have like some sort of hatred that's built up over time. And I think maybe they found now that oh this is a great place to go and display my hatred and go make my voice heard. Does it feel like everyone there is brainwashed? Like some kind of mob mentality stuff? Because that's why I would be scared personally to go to one is that like if someone's like Jack Neil like get that guy like they're all just like they all want to attack something and they all kind of have this hive mind. Uh so that is what would freak me out. Yeah. >> Yeah. There's definitely a sense of mob mentality. For instance, just the past few months before all this started happening with the daycarees, I would go to these protests and instantly within within minutes, I'd be targeted by these protesters. They'd come in, block me out of the protest in New York City. My last one I did, uh they actually arrested two men for uh coming up and attacking me during the live stream. One guy came through, hit my camera, took it off, took the camera off the the gimbal, nearly broke the microphone, and then another man came and targeted me and they arrested him. And so there is that sense of bomb mentality where you whole groups of people will just come after you, and why why are they coming after me and why why aren't they going after the other reporter that's there? Is it just because I'm Nick Shirley and I have a presence and I've uh spoken and given a briefing to the president or is it because they think I'm like uh conservative that I'm against everything that they're about when I just go there to give them the opportunity to voice their opinion? >> I feel for you in some ways that uh this event as good as it's been for putting you on the map. Um, it seems like a lot of what you liked was doing on the ground investigative journalism of like some of these protests and some of these events. Like, does that sadden you a bit that you're probably not able to do that for a while? Yeah. Well, I think my name's been like painted inside of these groups. For instance, in Minnesota right now, they have me on light poles saying, "If you see Nick Shirley, tell me." That's not cool. I would never say that to you. I would never say that to anybody. So, how does somebody get to the point where they think they can tell somebody go and kill themsel? Do they not realize that I also have a a family that loves me? I have a mom, I have a dad, I have siblings, I have friends, I have grandparents, I have a lot more to accomplish and to live for in this life, but meanwhile, they'd rather have me killed. >> January 6, your mission is the mission after January 6th. >> Yeah, the mission's after January 6. So, I'm a member of the Church of Jesus Christ of Latter-day Saints. When you turn 18, you're given the option to go on a mission. And uh you go and you teach people about the religion. Um you teach them about the gospel of Jesus Christ. You teach them about the restoration of the church in which we believe in. And uh your mission is to go and teach people just like the disciples in the Bible. Like go teach them about the gospel and baptize people. And uh so you're given that option when you're 18. I decided I was not going. I told all my friends started going and I was like, I am not going. Like I don't even know if I believe in this. Why would I go and do something for two years if I'm passionate about this other YouTube stuff? Like why would I? And it was during COVID time. So I'm like, why would I do that? And like I don't even know if it's legit or not. And so I'm not going my I remember telling my dad that and like the disappointment in my dad's face when I told him I wasn't going to go on a mission. I was like, "Oh my gosh." I think that's one of the only times like I've actually like cried of like disappointing someone. I remember like getting in my car after and I just started crying. I was like, "Oh, I felt so bad cuz I knew it meant a lot to my parents." But at >> what did that conversation look like? >> Um I was going about getting ready to go out to hang out with my friends for the night. And my dad would ask me, he's like, "Nick, have you thought about going on a mission?" I'm like, "Yeah, and I'm not going like I don't want to go." And my dad tried his best to be supportive and he didn't like push me on it cuz you can't force him to do something you don't want to do. And my dad just like looked at me. I could just kind of see the sense of disappointment in his face and uh like he still loved me and everything but yeah I could tell I just kind of disappointed my dad and I was like oh crap. Like I felt really really bad. >> Did your brothers go on their missions? >> Yes, both my brothers did. How common is it for Mormons to go on missions? 98% of you guys are like 60% like what would you say? >> It's very common. Like out of my friend group, out of the let's say out of the my 10 friends, eight of them probably went. >> But your family's like particularly religious. So >> yeah. So like, but like growing up in high school, my group, my friend group of 10 people, it was just me and one other friend that didn't go. And I didn't want to go. I didn't know if I believed in it. 2 years was a long time to commit to. And I moved to LA. I did did that. And my YouTube was doing good and everything, but did LA make you feel like you needed to find God a bit? A little bit, but it more so just like like going along those principles of like no not drinking alcohol or not even principles, it's just like those morals of not drinking alcohol. I saw kind of like, oh, like religion is good, like having strong morals is good, having strict discipline for yourself is good. It didn't make me sway more towards the church or anything like that. But I was like, oh, like believing in something is good. Like having a firm foundation is good. But going along continuing the lines of that, I kept doing collaborations with YouTubers. I would do collaborations with YouTubers like with tens of millions subscribers like Tanner Fox, Air Rack and I or I even got invited to Logan Paul's house by uh by Mike and I would show my analytics and they would look at it and they're like what is going on? Like you have all these like your your thumbnails really good, your clickthrough rate's really good, your retention rate's really good. Why isn't your channel blowing up? And I was like I don't know. Like am I getting censored? They're like no. They're like, "Well, maybe." And I'm like, "What is happening?" And so, my channel wasn't taking off, even though I was doing everything right and the topics were good and anybody else would have done it would have gotten hundreds of thousands of sub of subscribers. Anyways, I moved to LA. I moved out of LA after that cuz I didn't like it. I moved to Florida with my brother to go do uh door sales. >> What' you sell? Solar or >> uh pest control? But I I so I moved to Cal Florida and I do door to door sales and I hated that. I was good at it but I hated it cuz you're knocking doors for 8 10 hours a day trying >> this in the summer. >> Yeah. >> Four months straight. That's all you do is literally knock doors Saturday. Yeah. I did pretty good. >> Um a lot of people make hundreds of thousands of dollars going out in the summer. So I didn't do a fraction of that. I think I made like 20 grand while I was doing the summer sales. I felt like uh maybe I should just give this mission thing a try. Like uh why why why why does it keep popping back up in my mind I should go? And so then I made a decision. I was like all right I'm going to go for 6 months and if I don't like it I'll leave. I'll go home. Like I I you're supposed to go for two years. Like when you submit your papers you're dedicating two years. I was like I'll like forego that and I'll just go for a six month trial mission. We'll see if I like it. And I end up going on my mission. It turns into be a great experience. since I learned Spanish, meet really awesome people and I have a great experience and had great uh experiences with just spiritually and with teaching people about Jesus Christ, seeing how really just giving people simple things such as praying. I remember praying with a little kid who had never said a prayer before and for his first time saying a prayer he just lights up and he just says it feels like he's flying cuz he had never given a prayer before and he had never felt that connection and he was a little kid. And so >> what prayer would you usually tell him? >> I believe prayer is a conversation just like me and you are having with your heavenly father. So you say it, you direct it towards your heavenly father god and you close it in the name of Jesus Christ and in between there you can say whatever you'd like. And so it's a conversation you're able to have just like you are with your father here on earth. >> And you're able to pray in Spanish. >> Yeah. I speak I speak fluent Spanish. >> Yeah. There's a thing about Mormons on missions learning Spanish or uh different languages super quick. What is with that? >> Yeah. I learned Spanish in about 6 weeks after >> the CI actually like studied this because they have to learn foreign languages uh like multiple foreign languages and they studied Mormons specifically to figure out like why they learned languages so quickly. Do you do you know what it is? Is there just like who taught you? You just learn on the ground. >> Yeah. So you go through like a training for like 6 weeks to learn the basics and then after that they just kind of they just throw you into church. >> Yeah. >> Okay. And so more like I could give you a spiritual instrument. I could give you also like a logical answer. Well, that would make more sense for people. But um also >> give me both. That's interesting. >> Yeah. So like a spiritual aspect it would be the gift of tongues which is a gift that you can receive. Um so the gift of tongues you can learn languages faster. And is it easy? No. Um and like are do you learn the language overnight? No. Um, but I think if you're working and you're aligned with God, like you're going to receive certain blessings. And that's one of those that can come to you is like the gift of tongues. And like I mission, there's lots of times where I had no idea what what I was like trying to say or if people could understand it and you leave the conversation, the lady like pours out her heart to you and you're like, "What do I say back to her?" And then you just open up your mouth and you just try your best to be sufficient and that resonates with the person. And um >> I think my grandma has it. Uh but she doesn't she doesn't know what language she's speaking, but it sounds ancient. Uh she's done it for me one time and I I didn't believe her and I was like that is a language. Uh but is it just Spanish for you or have you ever spoke any other languages? >> Um Spanish I speak like that's all I've learned. I learned I tried to learn Portuguese for a little bit for a YouTube video to go in Brazil. >> It was super tough. But so on a spiritual way that I would say it's the gift of tongues and then like logically speaking well if your only option is to be speaking Spanish for 2 years you're going to figure it out like you have to. So I was with that with along with that I'm studying Spanish for an hour every single day at least an hour every single day. All the conversations I'm having are in Spanish. Everything I'm hearing is in Spanish. Everything I'm reading is in Spanish. And so like you're going to have to learn Spanish very quickly. And uh >> do you have friends that have like learned multiple languages from their missions? >> Yeah. Well, >> did they say it's more of a spiritual thing? >> Uh for instance, I had a friend who went to South Korea and it took him nearly a year to learn Korean and I Spanish is an easier language to learn. And so it happens in different forms for a lot of people. I had a friend who went to um who learned Russian on his and I had another friend who learned French in his. So you kind of everyone learns differently and it takes longer for each person. Like some kids on the mission it would take them like for me I got lucky cuz I kind of already knew a little bit of Spanish from when I was younger. Took me about 6 weeks to be able to start having conversations. For other people it took them 6 months. And so there's not like a set time when you're going to learn it. but I kind of want to go back. So, I do that mission and uh I came up with a plan while I'm on my mission of exactly what I'm going to do when I get back. I created like a 5-year plan and I created like a plan specifically, super in-depth plan for the first year of me getting back, exactly what kind of content I wanted to get into. I remember like I had the thumbnail idea already set in my brain as far as like, okay, it's going to be a clip from the video and it's going to be the lettering in the yellow with the yellow box. It's like I I I know that looks aesthetically pleasing. And so I came up with this plan exactly of how I was going to do YouTube when I came back and uh go down to the southern border to see what was happening. And since I spoke Spanish, I was able to interview the people and interview people coming across the border. And so then from there, I just kept doing more and more videos on the topics of migration. And and now it's kind of my topics have gotten wider as to what I can cover, whether it be the prisons in El Salvador, the FLLAS in Brazil, or the fraud happening in the United States or the fentanyl crisis. Now my my filter of content I can funnel into my channel is a lot bigger and my success took place. I think a lot of it was like just God's timing and what he wanted for me. cuz if I would have blown up when I was in LA, who knows where I'd be now. And so, and who knows where I'll be in 10 years. So, I just take everything for what it is and try not to rush and to really just have um really just have like the faith that what I'm doing is what I'm supposed to be doing on Earth right now. >> Do you enjoy it? >> Yeah, I do enjoy it. I do enjoy it. Like, this is all that revolves around my I think over the past two years I've probably gone to like 10 outings with friends. Like I don't like I don't focus too much on the friend aspects or going too much on dates or anything like that. Like I'm focused on what I'm doing right now. >> Would you say Mr. Beast is your biggest inspiration? >> Definitely one of them. I think like in he did a podcast where he talks about like well if my mental health was a priority I wouldn't be here. I feel that same way in a lot of ways where I really don't really believe in that sort of stuff. I think um like to be successful, you just have to be a bit delusional to the point where if you aren't feeling good mentally, you just have to keep going and keep pushing through and try not to focus too much on that sort of stuff. of all the stuff you've seen like the prisons in El Salvador, the fraud in Minnesota, uh the border crisis, the Did you go to Appalachia for No, no, no. You went to Pennsylvania for the opioid epidemic, right? >> Um yes, I've been to Pennsylvania and I've also been up to like West Virginia as well, but not for that. That was more like a going But that was for a separate video. It wasn't about the opioids. if like of all these issues like what do you see as the most like heartbreaking issue in America right now or maybe the number one like this shouldn't be happening? Like I know your whole thing is fraud right now, but like do you feel like that's the biggest thing you've uncovered? It's the biggest thing I've uncovered because no one's been able to put a spotlight on it like I did. I think the most tragic thing happening in the United States right now is if you go to any major city in the United States, there's a location where people will be strung out on fentanyl and that should not be happening. I remember homeless people back in when I was younger. I mean, you give them $10 and you think, okay, well, hopefully they're going to go get some food for the night. But now these people are all strung out on fentanyl and they can't they can't even they can't even get to the point to ask you for money cuz they're so strung out on this drug. Like they're not even focused on getting the dollars. They're more focused on how they're going to get their next hit. Whether it be from stealing or whether it be from getting the money from their friend or a lot of these people like in California, they receive money. They receive an allowance from the government and they're able to use that to funnel to get all their drugs. And so I think that's the saddest thing to know that people are dying on the streets of America. Like there's not one good there's not one city in America where it's in a better position where it was than 10 years ago. Maybe maybe there's one or two cities. Have you ever given anyone Narcan? >> I carry it with me in my car. I haven't given one any one p personally yet. I had to go and run and grab somebody to give it to somebody in the streets of San Francisco cuz someone overdosed. And uh this guy overdosed on drugs and he was in a group of about 10 people. He overdosed. Everyone's freaking out. And so I run over to this person to go grab him to give him like, "Hey, get the Narcan. This guy's dying." And they run over, give the guy Narcan, revive him, get him back to life. And that happens every single day. I mean, in California, they will literally give you the paraphernalia to do the drugs. They'll even give you the monthly allowance pretty much to be able to buy your drugs and then they'll give you the Narcan to bring you back to life. What are these people like? Yeah, a lot of these people they were they either got in a medical they either got in an accident and they were given uh opioids or they were been longtime heroin addicts. And a lot of these people are could be highly functioning people. Then they take fentanyl which is 50 times more powerful than heroin. And after that their entire life is them on the street trying to get their next hit. It's a really sad cycle. Yeah. They're usually prescribed like hydrocodone acetaminophen and then they can't really afford that anymore and they want something stronger. So, I don't know. They try heroin or maybe they try uh fentanyl and it's all just because you're saying they got injured at work or something like that. >> Yeah, they Well, they got injured in a car accident or maybe they just wanted to do opioids. Like I have a friend in Washington. And I grew up in I lived in Washington for a little bit and I remember in 8th grade my friend named Josiah his dad gets shoulder surgery and then he he starts stealing his dad's opioids pills and that happened in 8th grade and to this day Josiah's still on drugs. One of my good buddies I grew up playing football with, he died from fentanyl overdose and they're just normal people who got hooked on drugs. A lot of these people are just normal people who got hooked on drugs. I think from like that high school I went to or the kids from my middle school. Um like over 10 kids have died from fentanyl. I was just talking to my friend uh this other week cuz he's still up there and yeah, a lot of the kids growing up in that middle school I went to. A lot of them have died from fentanyl overdose. I can't remember the statistic on this, but I think more people died of opioids in 2020 than of CO, which is really crazy just how overstated one was, but how underaddressed this issue is. You think there's any corruption with like the pharmaceutical industry or does that not interest you too much? >> Oh, there's a lot. I mean, how come healthcare is so expensive here in the United States? And how can people just get prescribed all these drugs? Yeah. I mean, we're giving our money to uh learing daycare center like or what is it? A quality lingaring center instead of getting like free healthcare. Is that something you'd support? I know you have fairly conservative views, but yeah, I think as long as there's a way to make sure that cuz I hear people in Canada, they're like all proud about their free healthcare, but then you talk to other people and they're like, well, we can't get our we can't get in for >> months for like emergencies. There definitely needs to be a way where healthcare is not so expensive. And I I'm not no healthcare genius by any means or but uh yeah, definitely needs to the prices need to get dropped down. Why why are drugs a lot cheaper in um Europe for instance than they are here in the United States. And uh what's interesting is you see Trump trying to cut down on a lot of those prices like for insulin or stuff like that. And people get mad at him. And it's the same group of people that get mad at him for deporting illegal migrants or the same people that got mad at him for capturing Maduro. Well, if he does something good, shouldn't we celebrate it? If if murder rates are down all across the United States as Trump's been deploying the National Guard to these cities, isn't that something that we should be happy about? Yeah. Trump could stop an alien invasion and I think the left would have problems with it. You know what I mean? They'd be like, "He killed an alien. It is weird like how much polarity there is between the two groups and like your case is something so objective in the way it was presented that just like immediately got politicized. Uh which is fascinating and you didn't see it as a political piece at all. No, like in my video I didn't reference um like I didn't make it a right or left issue. I literally just talked about fraud. I and I said I think at the very end of the video I just said like we need to hold corrupt politicians accountable. I didn't say we need to hold corrupt Democrat politicians accountable. I just said politicians. I'm going to give you a name of someone and you give me one word to describe each person. So Na Boulli >> legendary >> Ilhan Omar >> fraudster >> Mr. Beast innovative clvicular >> interesting I don't know much about him. >> Hassan call me. >> He's came after me and I'm like dude I just exposed fraud. >> I was like you looked like you're about to say something nice and you're like call me. Uh, Piers Morgan, >> hypocrite. I think I'm going on him tomorrow actually. But I don't understand him. Like he he'll say sometimes you'll think he's uh right or then he'll say other stuff. I don't think he totally knows what he believes in either. Jacob Frey, weak. Tim Waltz, delusional, Nick Fentes, captivating. I think he captivates a lot of people. That's why you see so many people interact this stuff. Donald Trump polarizing, but at the same time, an icon. Charlie Kirk, Martyr, >> Gavin Newsome, >> Sly. David, legend. Nice. Uh, I wanted to ask, so your mom, she she actually just uh texted you. She's like, "We got to get out of here for your other interview." So she >> No, we're good. We're good. We're good. But she seems very involved with your stuff. Uh, and it's strange because in many of the videos that you featured it featured her in, like you guys are in fairly dangerous situations. Like at what point did you realize you were protecting your family like over yourself and like you weren't a kid anymore in that way? Last June, me and my mom got jumped in Minnesota. a group of Somalians stole my camera. We did get it back from them, but for about 5 10 minutes, we were fighting with these Somalians to get the camera back. My mom had to step in. She didn't have to. She stepped in on her own. And then at that point, I realized, okay, this is probably getting a little too much to be bringing my mom on a lot of these locations where I have to be very uh like I didn't want to bring her in that video initially cuz I knew it would be be bad, but she wanted to come. And then after this video, you've just like all my family, they've all taken like certain caution or like I've >> Do you feel like you're keeping them safe now? >> No. No. >> What do you mean by that? Like am I keeping them safe? >> I guess like if I took my mom to some dangerous event, like I'd feel like I'm the one protecting her. But like when I was younger, it wouldn't feel like that, you know, cuz it's like you know more about this stuff. You're kind of the person putting in the reps on like where do we go? who's dangerous? Like, is this a bad situation? You know? Oh, well, in that case, like, yes. Like, I'm not going to bring my mom with me. If I go back uh if I go back and do a video, I'm not going to bring her with me while I filmed a video or if I go to another protest, I'm not going to bring my mom. But if I go to Has she been a target of people? >> Yeah. >> Really? >> Like now, cuz she's my mom, people affiliate affiliate her with me. Has she got any weird messages? >> I don't know. I literally don't think she's like understands how the internet works completely. Like she doesn't even know like message like she gets message requests on Instagram. Like sometimes I'll grab her phone. I'll like check to see what people are saying and she has no idea. So that's great. But people love her on the internet, too. >> What's one thing your mom taught you? >> She's very resilient and she's a hustler. mom, like like I said earlier in this podcast, like growing up, she would like clean houses so that way that way we could get the new shoes we wanted or that way we could go and do our go on trips when we were younger. She's just a hustler from um whether it would be cleaning daycares or watching her friends like um helping her parents like she was she's a hustler and she gets things done. Like if I say, "Mom, okay, will you plan this trip? I need to go to here, here, here, and there." um she might do it in an unconventional way, whether it be on a notebook or or on a laptop because she has a hard time with technology, but she'll she will get the job done. Like my mom is a hustler and she always has been. Like if she needs $500 by the end of the day, she will go find work to get $500 in cash if she wants to. Like I had that same mindset. So, like if I needed $500, I would hurry and go to Home Depot, go buy myself a power washer, and I I can say this cuz I've done it before. Go buy myself a power washer and go to a neighborhood and power wash houses for a day and I'll go make $500. But I think like being a YouTuber inherently, like you're a hustler, like especially in those early phases phases of YouTube, like you will do anything to get it done. Like I have a lot of respect for YouTubers who go for years and years and years and then they uh without receiving any recognition and then they get it because YouTube for instance YouTube like the the percentage of succeeding on YouTube is so low. Like people think it's easy but no it is so hard. So many sleepless nights, so many flights that you didn't want have to get on to. so many hotels you don't want to have to sleep in in so many situations where you're just like literally risking everything for that YouTube video for the hope for it to get seen in the hope for you to be able to make a profit on it. It took me 5 years to make a profit on YouTube. What's the scrappiest thing you've done to get in a room you weren't supposed to be in or meet with someone who usually wouldn't meet with you? I think when I was younger, I think I was about 16 years old, I snuck into Jake Paul's wedding and I remember literally hopping the fence at his wedding to go in and interview and go speak with people and to go make that YouTube video. And so that was very scrappy on me to go and do that. and uh then like hopping in the party buses and then going to the party and sneaking in from the back door, the red carpet and then uh just to make that YouTube video. Like I would do anything at that time period to make something happen. >> It's interesting. Anything What's your advice for uh reaching out to people? Because I mean this is just my genuine curiosity, but I think it's something so many people struggle with like how do you get someone to respond to your message? >> Yeah. Yeah. Well, a lot of people will message me and they'll say, "Hey, Nick. Um, I would love to work with you." Uh, and they leave it at that. When I was younger, I remember I was able to do a collab with, for instance, with Eric when he was couch surfing the Logan Paul's couches. Like, and the reason why I was able to get responses from people like him or even to this day, I'm able to get responses from people is because I try my best to provide as much value as possible within that message. Similar to what David did with me. A lot of people have messaged me about the fraud in Minnesota, but nobody messaged me with the value of what was actually happening. So, the most important thing when you reach out to somebody is a lot of people are just looking for ideas and they're looking for value within that text message. So, if somebody wants to come on your podcast or somebody wants to reach out to me, say, "Hey, my name's Nick. I've been doing so and so for this many years. I think I could provide a lot of value making clips for your YouTube channel. here's actually five clips I made for your YouTube channel. Let me know what you think. I'd be much more inclined to respond to that than somebody says, "Hey, um, I'm looking into start making clips for YouTube. Can I start making clips for your channel?" >> Yeah. And then they ask for money or something like >> Yeah. But horrible way to go about it. >> Provide the value firsthand. >> Yeah. I told people uh I say it on the podcast regularly to uh just like make sure the right people actually do it. It's like make me a clip page and go viral and then you're hired, you know, like uh because why else why would you want a clipper whose videos aren't doing well? It's kind of like I don't know why these some of these companies they hire people in marketing uh that don't have followers on social media. It's like what's their credentials? You know what I mean? Um >> like on my YouTube channel right now, I have some guys who make shorts for me and it's all their payout is completely revenue based. So if the shorts perform good, they make money. They don't they don't. So like give me an example of something you do now. So like if you wanted to who's someone relative to Elon like if you wanted to get an interview with uh Sam Alman tomorrow what would you do? So I'd either message him directly or I would find somebody within his inner circle. I'd send him a message say hey my name is Nick Shirley. Here's what I've done. I'm looking into AI. This is my uh my following. These are people I've interviewed in the past. Um, I'm willing to make it work. If he has 10 30 minutes, uh, I I can be there. You give me a time and place and I am there. >> You say 10 30 minutes. That's interesting. And kind of the hope is that it might be longer. >> Yeah. I'd say, "Hey, I'll give you just give me 10 30 minutes cuz if that's all they're going to give me, great. And if we start meeting and have a conversation, they gain that mutual respect for me, then they'll go longer if needed." And so, yeah, it just be something as simple as that. Um, and I'd send proof of the people who I've interviewed and send proof of my channel, proof of my work, and then provide as much value as possible. Say, I really want to talk about what you guys are doing in in AI and how it can be innovative or potentially some of the the harms that could come from AI and I'd love to help you reach a new demographic of people. What do you think makes the prison system in El Salvador so interesting? because you went to the mega prison to interview like the deadliest prisoners in El Salvador. Uh you talked about that a little bit, but why is that system so interesting? Because they went out and rounded up any gangster that was gang affiliated at that time in El Salvador like in 2015. It was the most dangerous country in the world, the murder murder capital of the world. And now El Salvador is the safest country in the Western Hemisphere. So they grabbed all of the people that were killing, extorting, and kidnapping people, making it so that grandmas couldn't go to the other side of their their street to go visit their grandchildren. And they put them all behind bars for life. They'll never leave. And um it's crazy to think that there's people in this world who have decided to become these gangsters, tattoo their entire bodies with their gang affiliation, and then for them to get caught and be held responsible. >> Are there different levels to the prison? Uh there's like a three- tier system, right? >> Yeah. So in El Salvador their prison system what you see is SECOT that's the maximum security prison for prisoners and that is specifically meant for gangsters and terrorists and then along that they have another prison called planet oio which means plan of no leisure and it's a prison system where these prisoners are being rehabilitated. They were not gang affiliated but maybe they sold drugs or maybe they had beat up women in the past but they weren't gang affiliated. And so they created this prison system and it's actually quite amazing because instead of here in the United States for instance prisoners just they're in the prison they're not doing much. They're not really providing producing or helping society in any way. But these prisoners inside of El Salvador because there's there's so many of them because how dangerous and how bad that country became. um these prisoners they've created this whole entire uh ecosystem where they create and eat all their own food and then they also help the country as well. They have these huge man manufacturing sites inside these prisons where uh these prisoners are creating the clothes for the children at school. They're making the clothes for the hospitals and then these projects they have inside El Salvador, for instance, they just opened up a new highway and a large portion of the of that construction's been done by the prisoners. And so inside of the prison, they're teaching these people how to become engineers. They're teaching them how to uh to fish. They're teaching them how to they farm as well, >> farm. They're teaching them how to manufacture clothing. And so, and each day these prisoners work, they get two days taken off their sentence. So, it's very cool to see a country that used to be the murder capital of the world. They're still very much a third world country, but they're a country who's breaking the cycle. And so, that's why when you ask me who Na Kelly is, he's he's legendary in the fact that he's liberated an entire nation. It's fascinating how they were able to do that. Um, two days are deducted from the sentence per shift. They all have like life sentences, right? So >> no, it's all different inside that prison. So in SECOT there, that's for the most dangerous, most violent terrorist. They're never leaving this other prison. Um for instance, one man who one man was a kidnapper actually, but he wasn't gang affiliated. Uh he had like a sentence of like 27 years and he's 60 years old. So each day he works, he's going to get that sentence reduced and uh then he'll have 13 years left essentially. I like that system because I feel like our prisons are pretty overflooded. Um, and these guys aren't really doing much in there, you know. >> No. And so I also think it's super funny when they attack uh Naive Kelly or they even attack me for going inside those prisons like well CBS and CNN has the same opportunity, but they're not going to do it. Um, because it's actually showing that like innovation and uh prosperity inside a country can be can happen in many different forms and different ways. And this man and NaB Kelly has done that. And if you go talk to the citizens in El Salvador, they feel liberated. Like they're so happy to be able to live a life where they can feel like they can go to the street across from them and go hang out with their family without being extorted from a gang. >> You were one of the last people to talk to Charlie Kirk before he was shot. Like what impact did that experience have on you? Did that inspire you? Do you think about that a lot? Charlie texted me the night before he died, asking me to come on his show. And I go on his show, he wasn't hosting it, but his uh co-host Andrew was. And then an hour later, I see Charlie Kirk got shot. And yeah, that impacted me a lot. I didn't watch all of Charlie's content. I wasn't a devout follower of Charlie Kirk. Um, I respected him similar to I have respect for you and how I have respect for other people and I I believe in what he stood up for. Do I believe all do I agree with everything he said? No. But do I think he should have been killed? Absolutely not. And so it affected me a lot because well, one, he reached out to me the day before he died. And then two, it also means, well, there is an attack. There's like some sort of warfare going on here inside the United States when they're celebrating the death of a man who was a father who had employees. He ran an organization and what was he standing up for? He's standing up for Christian values. He's standing up for the Constitution and they want to kill him and they want to celebrate his death. it m it made you really concerned for the state of our country and also makes you want to fight even harder for what's happening and uh so it impacted me a lot and I remember I don't think any of us had at least for me I had never been impacted by that way of a death of anybody else besides a family member just knowing that like there are people out here who literally want to kill you and so um yeah it's a tragedy it's it's horrible. Have you had any situations where you feared for your life like that? To the point where I thought I was going to get shot. No. And I hope that I never do. >> Mhm. >> But just the fact that I have to have security with me right now and when I travel and stuff, I mean, you never know what's going to happen. >> Who advised you to get security originally? Yeah, a lot of people advised me to get security and um Officer Tatum, if you know him, Tatum, he has like a company called Black Line Guardian. He has a nonprofit side to it where people can then go and donate for that security. And so, >> uh people have donated to that for me to be able to have security, which has been super helpful because obviously it's thousands of dollars a day and a YouTuber. You don't make that much money to be be able to have that. So, right now I've been able to have security just from that fund alone, from people donating. Yeah, the trolley thing was uh that was rough. I I think it's a shame that so many people even had to watch that video uh for their like own mental perspective, but the reaction to it, it's just absurdity, the reaction to some people's things. So, Charlie reached out to you originally for the border crisis. I think it was around like the Antifa stuff in September. So, he had just reached out because he had saw me on Fox News a few hours before and he just texted me, asked me if I could go on his show and talk about it. I'm not too well informed about Antifa and what they're up to. Uh, but I am confused about the border crisis. Like, as someone who's been boots on the ground, like what is the border like in 2026? Have you been there recently? I went there. I haven't been there in 2026, but I went there in 2025 and completely and drastically different than what it was like underneath Joe Biden. I remember going to the border. For instance, in this border town uh called Hakumba in California, you would drive down the border and you would see people walking across putting themselves in single file line for border patrol to pick them up, take them to San Diego, and then release them the next day on the streets at a bus stop to take them to the airport. And that would happen every single day in multiple locations across the United States in Arizona, California, and Texas. And so, and why wasn't anybody stopping that? I mean, children were being trafficked over the border. Women were being, you know what, by the cartel? Why weren't why weren't these protesters advocating for the human rights of these people? Like, why weren't they saying like, "No, like, stop it. These people are being trafficked." Where were the protesters then? Like, why? Why do we let that happen? >> So, you're saying people are being trafficked across the border. What does that mean exactly? >> Yeah. For instance, I spoke with a little I spoke with a teenage girl who was, I think, 16 or 17year-old from the Dominican Republic who came here all by herself. And in order to get into the United States, you had to pay the cartel uh a fee to get in. You weren't people weren't just walking across the border. They were getting trafficked over by the cartel. when people from China who are leaving communist China are making it all the way to our southern border and then there's instructions instructions in Chinese for them for them where to go after that should make everyone concerned that there are Chinese men at the border oh lots of them why who knows the same people there's also groups like WhatsApp groups where these uh Chinese like in infiltrated as well and they're taking photos of like the Navy and what they're doing along the coast of uh San Diego and that's been all well documented and nobody knows where these Chinese people are now inside the United States. >> So it's not just Hispanic people at the border. >> No, this border crisis was something that no one has ever seen before here inside the United States. Typically the migration was from uh South and Central America. This time we had people coming all the way from Marotania. People from all across the world were coming. And so like for instance the case of Somalians. Yeah. A lot of Somalians came in as well. They knew that people could just come inside the United States >> through the Mexican-American border. >> Yes. >> Huh. And even the Biden administration was flying people into the United States to hide the fact that they were that that was happening at the airport at that time. Uh these people they had they could go through TSA without even having documentation. They would have a I remember watching one time Homeland Security move a family of a bunch of little Latina kids through the airport and I saw that at San Antonio and so they were just facil like the US government was facilitating human trafficking all across the country and now you're seeing people get so upset about them but meanwhile they've closed the border more less people are being trafficked over the border less people are being you know what by the cartel and it's a very humane things that the Trump administration has done as far as stopping the border crisis. >> What else are you most interested in when it comes to fraud? Uh you have your next video coming out which we talked about uh with the transportation issue, but what do you think is uh maybe you don't have to say specifics, but areas that you're looking into? Yeah, I think the welfare fraud across the United States, that's like a huge issue because millions and billions of dollars get sent to these groups and these organizations, the nonprofit side of things as far as them receiving grants and these funds for the government. There's a whole lot of corruption in the United States. And we're seeing that um fraud is in all 50 states essentially. like in like in Hawaii they've been spending I don't know if the numbers in the billions but I know they've been spending years and years trying to make like a like their own like a railway but hasn't come nothing nothing's come about it same thing with California they spending years and years and billions of dollars on this train that's supposed to go from LA to LAX and LA to San Francisco and like no tracks have hardly even been laid >> do you think there's fraud in social security >> yeah for sure I I mean, do me and you even know where all of our money even social security goes to? Like, do we like is it I think it's kind of weird that every year me and you put money towards social security and we won't receive it until we're like 65 or something like that. And uh I think it'd be better if they just put it all into the S&P and we were able to collect money into the stock market instead they put into social security and we don't really know where that money goes. >> Do you think you'll have another moment as big as this Minnesota fraud case? like of something you uncover. Are you still looking for that? Do you think you found it? This video will be hard to top because it is obviously like most feed video on X, but I'll continue covering other stuff that will get great conversation across the country and across the world. I mean, people in Sweden are talking about this video and they're like, "This is happening in our country and people in the UK are talking about, oh wait, these Somalians are committing fraud inside of our cities, too." But what's the big one? Do you think it's going to be welfare food stamps? Like like what do I think is going to be the biggest fraud? Like the in number wise? I think the welfare fraud's pretty pretty huge. >> That's the one. Interesting. Where do you What state do you think you're going to go to? >> Uh probably California. I think California will be next on the radar. There's some other small states, too, that I'm going to go go to as well. >> Do you think there was fraud in the LA wildfires? Yeah, there was definitely a lot of the fraud happening and uh there's a guy down there in California who's exposed a lot of it and so I actually want to reach out to him and speak to him about some of that fraud because um I think they're still having a hard time even getting the homes constructed down there. What do you think happened? Like where would the fraud like the money they got for the fires to rebuild? Like it just never Yeah. I went to go and try to make a video on the fires, but I couldn't even get in because they had the National Guard outside making it so people couldn't get into the >> fire zones. They're probably trying to keep you out. My uh my girlfriend, she she went to a We live right by it, right by the Palisades. Uh, and she we live near that area of the Palisades and she tried to go to a Starbucks. Uh, and they were like just like border patrol or not border patrol. Uh, like the agents keeping her out like, "Hey, you can't be in this area. Like you have to get out of here right now." Um, that's interesting that that would Does that happen to you a lot? Like people just won't let you in the places. That case was different. Um, now like in these protests and stuff, they won't even let me get into their protest a lot of these times recently. >> What government program do you think hasn't been corrupt or abused? >> I don't think there is one that hasn't been over time. Like I think people, especially with the government, like the goal for a lot of companies is to receive a money from the government because they know that that's more money. like a goal for a land owner like they would love to have their piece of property be bought bought by the government because they able to spend whatever price and the government are great at overspending. >> Has anyone like given you a counterargument to what you've discovered that has been like oh that's interesting like I no I mean literally the HHS froze all the funding and not a single one of those daycarees has been able to prove that they're a legitimate business. A lot of the daycarees, I think five out of the seven daycarees I went to were also involved in the feeding our future scam. But it is funny to see like how people come out of the woodworks to try and paint you for anything to try and discredit you. And it's like no, like what I did was truthful. Like good luck, >> Nick. after everything you've exposed, uh, civilian fraud, uncontrolled violence, like the dark underbelly of government corruption, like what hope do you have left for America? And do you have any hope left? Yeah, I do have lots of hope for America cuz I think my generation, I think a lot of us just want to have the same opportunities that our parents and grandparents had. And now we're coming out and we're talking about things and we're standing up for ourselves the best way we can and by creating these conversations or at least I feel like I'm doing my part in trying to hold the government accountable to show people the realities of the world for what they are and to at the end of the day try to make America a more efficient place in a way that I can I'm 23 years old. I can't go in inside the government. I can't cut the contract or I can't tell the president to do this. I can't tell him to do that, but I can create conversation which creates change. I have hope that what's happening in the country is for our benefit. I I hope that President Trump's working for each and every single American. I hope that less and less fraud will be taking place. After seeing everything uh that you've seen, talking with many different types of people, what's one belief about the world that you had that you no longer hold? It's a good question. I would still believe there's more good people than bad people in the world, but now I do believe for certain that there is brainwashing. Tell me more. I remember when I was younger, I thought I remember hearing about people being brainwashed by uh religion or cults and stuff. I remember like how does that what does that mean? Like brainwashing. But now we're seeing real brainwashing in real time where people will support fraud or people will support anything. Well, where people will support, for instance, on the topic of immigration, they will support somebody who's here illegally. The words illegal or the capturing of Maduro. Joe Biden also wanted to capture Maduro. He had a $25 million bounty. Trump goes and does it and then the Democrats are coming after him for saying it's bad. And so brainwashing is very real. I've seen it firsthand. People, you question their logic and they they can't they they can't come up with a sentence because that brainwashing that they had actually never provided them proof, but they were told and they believed it. And now they and they can't think of other ways to look at things from a different perspective because of the brainwashing that's taken place. What advice do you have to Jinzy to avoid brainwashing? Like question everything. Like don't take my word for it. If you think if you want to question it, go ahead and question it. Do your own due diligence. I'll do the same thing. And so like people get mad at people on the internet for questioning religions or questioning groups. Well, you should be able to question and do your own critical thinking. Most things in this world come down to common sense. And if you can't come up with a common sense reason for things, question it. What's the best piece of advice you've ever received? >> Follow your gut. Follow your instincts. There's a reason you're receiving those instincts. That's God working with inside of you to tell you to go out and do something. So follow what you believe and follow what the feelings inside of you cuz those that's God working through you to propel you to where you he wants you to go. So if you have a thought that's been lingering for months, years, and you haven't done it, well, there's a reason why it keeps coming back. Like follow your gut, follow your instincts. That's God showing you who he wants you to be. >> When is the most important time in your life that you followed your gut and it worked out well? I could have quit YouTube years ago. For some reason, I had that gut feeling that I still needed to make videos. I still needed to do I had that there was a reason why I couldn't ever get the topic of going on YouTube doing YouTube out of my head. And now you're seeing it now where um I've been used and I've been used my abilities have been used to propel me. My my abilities and my experiences that God gave me have helped me to be get to the point where I'm at now. So he used me in a way that I didn't know at that time to then get to where I'm at now. And then he'll continue to use me for years and years until I die. If you could make any video in the world right now, what would it be? I was thinking about this because somebody else asked me that question the other day and I didn't really have a good answer for it and I said I would like to interview people like Elon Musk, Trump, Naib Kelly. Um I think there's this guy in Africa right now who's like saving kids from slavery and he gets them like on a boat and I've seen a few Tik Toks. I think that would be a very interesting video. He's saving kids from slavery on a boat. >> Yeah, that's interesting. like he's like uh going out I can't I can't remember exactly what country but he's doing some he's doing something along the lines of that. I think go doing that would be interesting or um making videos in certain like parts of the world that are kind of untouched would be also very interesting as well. Do you think you'll go to North Korea? >> I would I would go to North Korea there. >> Yeah, I would. I'd probably have to run the marathon that they make so people can run it. Like you know that marathon they hold. They do like a marathon every now and then. There's this YouTuber from the UK. His name was like a Harry I believe. And he did a video in North Korea and he had to run a marathon to do it. It was pretty funny. And they let you film there? >> Yeah, he he was able to film that. Well, everything's very controlled there. I don't know if they'd let me go because I interviewed a lady who escaped North Korea. But yeah, I have other videos I want to film as well. >> So, your dream person right now is Elon and Trump? >> Yeah, I think that'd be cool to interview one of those two guys cuz that one's the richest person in the world and the next person's the most powerful person in the world. >> What was like so impressive about Trump to you? Like uh there's an experience of like watching people obviously and then like meeting them in person. Um like what did you notice about him? >> Yeah, he knows how to take command of the entire room. and he gives everyone the option to voice their opinion. He takes in advice from or advice and questions from everyone. When I went and did the round table, there was I think there was like 12 of us. He listened and let all of us speak for 2 minutes or 5 minutes and he took in that and he took took down information and he let everyone speak and he was able to listen to everybody. And I've heard that from multiple people like he will take opinions from everybody and he wants to know what other people think. crazy like genetics on that guy like his ability to just go go at that age. Uh I I will ask you just my own curiosity. I'll probably in the interview around here but um you think AI is going to kill us all. Have you seen that stuff? Will it kill us all? I don't know. Do should we like limit the AI here in the United States? It's hard because other countries will get ahead of us if we do. So I don't look into it. I don't think about it too much to be honest about AI. It'd be interesting to see a video of you on that. I went into that topic pretty recently. I had like a AI safety guy like he invented the term like AI safety and like has been researching it for 13 years. He's like of course it's going to kill us all. And I'm like gez. But yeah, it is the issue of other countries building it. And it needs to be like some nuclear war type agreement I think you know like everyone has some agreement that we don't launch nukes on each other you know. >> Yeah. Like should it should AI control the military? No. But also, can't they just like pull the plug on the on the super machines? Well, once it's super intelligent, I don't think we can pull the plug. Well, can't it has to get power from somewhere? So, can't you just like unplug it? If it's smart, it's like kind of like uh how would I explain this? Imagine if a dog made a human and the dog was like, "Oh, we're just going to like bite it." Do you know what I mean? Yeah. >> Or uh like an ant made a lion and it's like, "Oh, we'll just like uh we'll all attack it. Like it'll be fine." Um like that's kind of like when something is super intelligent by definition, it's way smarter than all of us combined. So we can't like it would take control of like every piece of technology if it wanted to in theory, but you can't really even predict what would happen because it's way smarter than we are. Uh it's something for you to deep dive into. I'd be interested to hear you talk uh to Elon about that like for a Gen Z perspective. But yeah, everyone uh this is the Jagno podcast. This is your guest Nick Shirley. Uh where can people find this hoodie? Show it to the cam. >> The quality layering hoodie can be found at surirleydeefense.com. >> shirleydefense.com. It's in all of my bios and descriptions. >> It looks comfy. Looks >> It is comfy. It's a thick hoodie. People are starting to get him this week. So, >> well, cool, man. >> Thank you. >> Thank you, Nick. Have a good one, brother. >> Awesome. \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_YouTube_Video_GfZneDlvPKQ_c61438f4.txt b/graphrag-ollama-config/input/youtube_YouTube_Video_GfZneDlvPKQ_c61438f4.txt new file mode 100644 index 0000000..c1cb560 --- /dev/null +++ b/graphrag-ollama-config/input/youtube_YouTube_Video_GfZneDlvPKQ_c61438f4.txt @@ -0,0 +1,9 @@ +# Source: YOUTUBE +# Title: YouTube Video GfZneDlvPKQ +# Channel: Unknown +# Video ID: GfZneDlvPKQ +# Ingested: 2026-01-12 17:04:12 +# Word Count: 4529 + +--- +hi everyone welcome to data analytics blog in this tutorial I'm going to be walking us through how to design and also build this project management dashboard analysis so the dashboard is about project management and I did my analysis based on Project type based on Project cost and project benefits by mod project how complex the project is the high or low and also based on our department and the face of issue of the project so now I'm going to start my analysis so first and foremost I did my dashboard this background design in PowerPoint as you all know so now we I'm going to show horse how I was able to do this particular background design so now just make sure that you launch your PowerPoint then I'm going to click on new style new slide delete this and then also remove this then now the next thing that I'm going to do now is to insert some shape so I'm going to be making use of shape and search go to insert and insert shape so I will make use of this rounded rectangular so I'm going to be inserting this so I'll just walk us through how I was able to do this to insert this so now I'm going to be filling my sheet no outline then color so I use a gradient feed so now I'm going to check okay let's check here so I made use of this this particular color and also this particular color so let's I will show you this is the color code for this this particular color this is the colored code for heat 25 125 and 130 and also the second one this is the color code in case you want to replicate the same thing so this is the color code for the lighter one so now the next thing now is to insert another rectangular shape to go to insert put you inserting color uh main shape so I'm going to make use of this so now I'm going to remove the outline no outline then my field will be white I want to make use of white okay so now I want to add effects so go to so I'm going to add a shadow effect we can add this a shadow effect okay let's check for glow effect I think I had a glow effect too the blue effects okay so I made use of this and I use this color as the blue effect I think I reduced the transparency yeah so now I can now duplicate this duplicate this so is what I did there is now a big something is not something that is difficult something that you can actually do so now resize that so now the next thing now is to insert another rectangular shape insert this so now I'm going to use this format for this thing go to format I'm going to format this then duplicate this again and bring it down so you can see what we did is not something that is big it's not a big thing so I'm going to duplicate this now so this so just make sure that you resize it to fit into what we want to do so now I'm going to duplicate one of these then bring it up with the kids will notice so now we are done with all the all of this so now the next thing now is to add our text box so now I have some things that I wrote here so what you just do is okay so now I'm going to insert the text we'll go to insert text box so let me just do copy and paste to what I've written here then I will just show us one of this then puts go back to this insert the text box so just bring it here so you can you can reduce your front you can change the front you can reduce it anyhow you want then you duplicate it sorry change it this give it customizes to your own so now you can write the rest so now the next thing is this this particular it's just a text box too so I copy this I'll copy this I'm going to paste it here so you just detect what I wrote here so here I'm going to duplicate this bring it here then write my dashboard title projects management dashboard so for these two for these two uh icon so just go to insert and make sure you have internet then go to icon go to Icon still loading so now I just choose I think I choose this and also one of this I think this so just inserts then bring it here so you can resize it then deselect it then you can now put it and here so here I'm going to change the color code to White and that's better so just make sure that everything is aligned so for this particular one so I downloaded it I just downloaded the project management logo you can add it so now I just go to insert go to picture then this I think that is this so now I resize it so just download any project management logo it doesn't have to be mine can I put it here so that was what I did here so you can just add everything then how to save I say just Ctrl a by highlighting comments highlight on one of the or what's it called one of the okay before then for you for this one not to be moving upper that you can actually group them together highlight it then go to formats or right click and click on group group so now if you want to save this just highlight everything here then right click and save as picture so you can just save it to anything that you want so I already have what I want to use so I just use this to for as an explanation so now we move to power bi to start our analysis to um just launch your power bi then import he I think is his CSV by so let me check and let's check no okay okay it's now download and okay so that's it so now we move it to our data is connecting so now we transform our data to check if we have the correct data set for each of the column so we do that in power query editor so now we have this so we'll just check if we add them in correct let's check this Co number okay I think everything is in what you need percentage let's change this if using the right for the this should be good let's check I think checks oh no not time not time let's check it let's check if it is dates okay um you will give it like this okay apparently this particular teacher this particular data I need to do some in the okay so I think we are done with this with this transformation nothing there is nothing much to do here so now we can load and apply choose and apply okay so our data is loading so now our data is loaded so it's time for us to bring our dashboard our a background design so just so for the new Power bi you can just right click format canvas then go to browse make sure that you know where you save your picture so this is my picture my design then bring it in reduce transparency and fitting you can fit or feel so now we start our analysis start our visualization so now the first thing now is I want to do all this kpi so now in this video so I'm going to show us how to make use of our filter so now this kpi so I'm going to bring introduce card so we can make use of the new card or the old card okay let's make use of the new card so now here you can open new pane so you know we have new power via interface so now we can start our analysis with this so now it's time for us to bring our so I'm going to bring this project benefit into Data so I have my project benefits here see loading so I have it here so now what I need to do now I want to filter for income generation from our project type if you check the data if you check the data you will see that under project type under projectile we have we have oh I think four different cost reduction income generation Pro process Improvement and work Capital Improvements so I want to make use of I'm going to these are the kpl I'm going to use so I want to look at the approach The Profit the profits for each of the project type so now I'm going to filter for project so now you open your filter here here if it's not ready so now you bring the project type into this place add it to this feed then you filter for the one that you want to make use of so I'm filtering for income generation so you can see now so it's time for me to start formatting my card so now I'm going to format the card so for the first thing that I'm going to do I'm going to remove the background remove the background again I'm going to what is remove the background from here then shape okay let's just leave it like that I don't need to do anything with the ship since I'm removing the background so now the next thing that I'm going to do is the call out to call out okay so I'm going to work on the call out change it to my preferred color so the color that I use so let's check that here I want to check for the color that I use here formats color value let me check for the color code that I use okay this is the color code I can copy it let's see if it's coffee then bring it here okay it's not copy let's go back and check it 1979e so 19 79 here that's the color code so now the next thing that I'm going to do now for the lighthouse I think he's okay we are cool with that again for the color remove the label I don't want to see the labor also because I already have my title so now we move to formatting card remove fee I don't need the fee I don't need the Border either so okay I need the Border because I want to add some accent but I like adding acid back to my did I add it to this let's check before we proceed because I want to repeatedly exactly what I did there okay no I didn't had it so I know where I added the assets back so I don't need to add to remove the Border so now we can bring so this is done now we can bring it here you can still format it just make sure that everything is aligned so now let's go back to our format go back to format and do something on the call out so you can change it to I think I used 30. you can change display you need okay let's take it to minions then to two significant Figo that I use so now so we are done with this okay let's go back and change this I use caliber for everything I did so you can also put it or we can leave it like this let's leave it like this just customize it the way you want your car to be your so now we copy and paste it three time for other visual so I'm bringing TCA also bring DCA so bring this here so now the next one is process improvement process Improvement I'm just going to go to my filter remember the filter that we did already then click this remove these and click on process improvements so same thing with these two go to filter they cost of reduction so now here I'm going to remove this and add this so now we are done with our first visualization which is our kpi so we are done with that so the next thing is the the composition tree so we are do we'll make use of the composition tree so this is decomposition tree and I will explain what the composition three or three uh visual so it's actually lets you visualize your data across multiple dimension in popular so what's the composition actually doesn't speak it's automatically aggregate data and enabled related into your dimension in any order so you can Google on that or watch video on their video on YouTube on the composition but that's what it actually does its lpu to uh aggregate data enable enable drilling down into your dimension like to allow you to visualize data across multiple Dimensions so now I want to visualize this I want to do my visualization I want to do my visualization using departments and sales I are going to add had add myself here that's completion so I want to visualize this average of completion that's what I want to do okay it has already aggregated it so now I want to visualize this based on let me check if it's Alexander okay let's check this I'm going to I need to do something on this erase this so let me change this to this so now so I want to explain this I'm going to explain it by department and also I'm going to explain it by face of the project where the projects which phase is the project in so now each department I'm going to visualize by Department explained by Department you can explain by difference in but based on my home the analysis I want to do using this data I'm going to visualize by face so you can just explain you can keep adding different things okay so now here I'm going to show by department just click on the button you can see now it has aggregated everything so now the next thing I'm going to show now is face again I'm going to do it by face so now you can see now that we have our decomposition trainer so now you can see you can start analyzing it so now here is average I don't want to add that off you can do some just formatting my picture so now the next thing now is to start doing my analysis so now I'm going to format go to formats don't need this again so go to format 23 so these are the so you can remove the background on you can switch it off so analysis mixture is in absolute 23 so you can do different thing if it is dense that you want it to be you can see it has changed or you want it to be just make sure that you just play around with it so they connect so I'm going to change the color to this change the color of the connector to this also the unselected one I'm going to change it to this so now my bath I'm going to change the color to I'm going to change the color to this let's check for my other color that you see let's check for the color that you see color so okay two triple E one that was what I use so now I'm going to use that to to you can customize these two what you want just to your home costume color okay let me check again I think I miss 2A okay to a that's 2v so that's the color that I use so I'm going to use the same color for these two so now the next thing that we are going to do just make sure that you customize this to what you want so the category label I'm going to use think I use this I used this for the particularly but let me see okay yes and custom like I said customizes to what you want so now the next thing now is okay creators you can remove Jesus try to values and change the value to black or any color you want to see and put it yeah So This Bar let's check I want to check something in the bar it is background color let's make use of this yes so now we are good to go so we are done with our decomposition trainer so the next thing now is our what's it called so you can just write projects by project completion rates so as the title let's check if you can put on the title so projects completion rates so this and this this so we are good to go so the next thing that we're going to do is our project by manager so how many ish how many projects per manager like project file manager so I'm going to use partial pie chart for that so click on your pie chart and come right here bring it here then the legend will be manager project manager then this will be projectile so now it's time to format format it so this let me just write projects formatting the project by manager so that's my title so you can just format it to your own color so you can customize this to yours so this okay so now the next thing I'm going to remove the legend I don't need the legend so this I'm going to use this the data label I'm going to use the the category and data unless format this this then let's change this front I want this front you can load it if you like so again I think we need to remove the let's remove the background switch it off so what's next our slides are yes our slicer I want this a slicer these you don't need to to differentiate like different slicer because I already added the category so this this this work can we use what can we use let me check the color that I use here okay I used to color here I used three color let's check our bar here slices let me check this particular color okay it's my color here so now let's try this 1978 true okay so here let's use this 19782 key can I add this to it this you can change it to this and yeah so now we are cool with this so we are good with this so we can now start this particular analysis which is the third or the last three parts so I'm going to be doing project benefits and cost by modes so I'm going to be using this cluster line and column charts drag it see my hand now this will be months changes to months then column y we can use costs as well then you can use this for this to be benefits or anyone who watches okay let's interchange them changes to this for okay then every should be for long I want it to be cooler who came from Netflix okay so let's do our future uh our formats so Legend try to text and change it to this thing here option I want you to be top Center foreign so what else can we do here then our grid line I don't need the grid line put it off just customize it to what you want so our column will be this let's use our column to the base for let's change it to this then our line will be this color our line will be this so what is I will do it okay so our y axis okay why actually remove this then the value let's change it to this black and then customize it to your home to our hex teaspoon this then the value to this black customizes to what you want good so we are good to go for this okay our title let's check our title foreign foreign by complexity how complex the project so I'm going to use a full name shot for that let me change it for nature with that to bring it here so category category will be complexity by this project type no okay okay complexity let's check complexity so this should be project type they will change the value to yeah okay so now let's format it I'm going to close this and of course because on the format so let's format it so now let's do some other formatting you can notice good so yeah you can write a make use of this then this foreign I'm cool with the option to the colors that she says at least this oh we can use this just use the one that you feel is okay so you can cook with this okay this let's draw shower category label so cool so this so the last chart is our what's it called that's our total project total projects and uh and the projects going and ongoing so I'm going to make use of card for this so I'm using the new card for this so inside go to insert then cards go to card so bring it here let's start the formatting okay add let's add to the projects now I'll be took up with it if you just know can't I can't let's see if it should count it yes this is what you want so now let's do the shape okay I think let's do this we are using this so now you can just okay let me be this like this you can reduce it you can work on it you can do anything with the new cards you can change the shape size call outs let's change it bring it here change it to our issue according to it then I think I reduce it to 30. which is cool then I change the color to this so our labeling change it to these two change the color to Black okay okay let me increase okay so now the next thing what else did I do here nothing so our layout you can change it to this you can leave it like that so now let's work on the cards removed it you don't remove border remove few I don't need fill then add a sense I like adding assets to make it pop out so change the color of the assets to this then change the position to top consider it has changed to top then you can increase the accent to whatever you want to increase it to so now we're done with this so I'm writing to top projects tutor projects so now I need to do the I want to extract the ongoing project and the project I stay on hold so I'm going to make use of this this particular this page so I'm going to add value so it will be my value will be okay that's projects project fees see project face let's check no project status is project status project status then I will make use of the filter bring the project starters to filter again so I'm going to feel tired bring it here so going projects which project is this let me remove this so this will be ongoing projects I'm going so now let's format let's use some formatting here bring it it tied to the need to okay you can reduce it okay so let's remove our background switch it off good color and the kings of this definitely let's use this so what is data labor Target label remove it collateral okay no notice this is um I'm going to use this okay so we can see do some formatting subtitude I need to format it again so here I'm going to copy then paste so I will use this for the projects on hold so go to filter then remove this and change it to project on hold so now is the percentage that we want to see so I'm going to change this I'm going to change this to percentage average let's change it to not count so I think I made a mistake that I need to change it to I need to change this to this so here we are making use of average of completion rates um raise it raise it we are using average of completion rate this they change it to average also this changes to average okay so it should be changing it back to project so I want to change this color to this crazy format and columns to change this to this so I will just formats and paste it so now we are good so now we are done with all our Visions now so the next thing I want to do is my slicer so now just bring slicer here then the first slicer is here so I will add here to the slicer Hardy so now format my slicer I want it as tie did I use tie the okay I use tie I don't need either so yeah remove tea then water slices values are changing into this so let me check what I use okay by changing it to this white so the background I'm changing into this notice this color so I'm changing it to this so now I have my slicer here the four slicer here I obviously so now the next one is so I'll just copy then paste twice for the order slicer so the next slicer should be the region let's add region and remove this removed here and add region so now the last slicer is the project status so we can check each of the status so we are adding this project status so we removed here and format this foreign so now we are done with our initialization we are done with our dashboard our report and I believe you have learned one or two things especially the future area from here so just make sure that you subscribe to my channel give me feedback from the from the uh comment section like comment and share make sure you share make sure you replicate this and also share on your page and tag me make sure you subscribe to my channel bye \ No newline at end of file diff --git a/graphrag-ollama-config/input/youtube_YouTube_Video_t_UwXi46iSA_2e245d1e.txt b/graphrag-ollama-config/input/youtube_YouTube_Video_t_UwXi46iSA_2e245d1e.txt new file mode 100644 index 0000000..574502e --- /dev/null +++ b/graphrag-ollama-config/input/youtube_YouTube_Video_t_UwXi46iSA_2e245d1e.txt @@ -0,0 +1,9 @@ +# Source: YOUTUBE +# Title: YouTube Video t-UwXi46iSA +# Channel: Unknown +# Video ID: t-UwXi46iSA +# Ingested: 2026-01-12 15:40:44 +# Word Count: 7116 + +--- +Welcome back. In this video, we'll learn about the Q&A feature in PowerBI. How to ask questions and get the answers in PowerBI in a very simple and layman language. Now, this is a feature about the EI in BI artificial intelligence in business intelligence. So, let's see in detail. So, first thing what we are going to do is we'll just pick up the data from one of the data source. I'll just click on Excel workbook. Name of the file is sample superers store data set. I'll pick up the data and from here I'll just pick up one sheet named as order sheet and then click on load button. Now Q&A feature is very interesting feature. It's there in PowerBI from I can say from 2020 year 2020 onwards but now AI is booming in the market but this was already present here. I'm talking about a feature it is AI visuals. The first AI visual which is Q&A feature here. Now let's see the data. First of all, we have this data set related to sales and profit. It's the custom information and all the details. We have date data type, ship date, order date. We have the textual data type given over here. We also have the geographical data type which talks about the country, city, state, postal code, region and so on. And we also have some textual data. Okay. And at the end of the screen, we have some numerical data which talks about sales, profit, quantity and discount. So I want to see the profit across all the dimensions. So first I want to find out how much is the profit received from each category, each subcategory, different cities and states, which customer gave how much profit and so on. So let's try to understand if you are very new to PowerBI, you do not have any information. How do you get the answers? I'll go to the report view option here. And previous version of PowerBI, we used to double click on this blank canvas and then used to get the answers. But now with the new feature you cannot double click and get the answers. We have to go to this drop-down select the option as Q&A option here and this will create one box. So preparing Q&A what it will do is it will give you a text box where you can type and ask a question and based on that it will give us the answer. So you see ask a question about your data and here in this box I can click and I can talk about profit. So when I say profit this is a name of the column in the data set profit by region. So when I say profit by region this will give me uh how much profit they have received based on different regions. This will take some time few seconds and it will give you the answers very simple language. So it says profit per region. I can clearly see at the data that west region has the maximum profit and central having the lowest profit. Now if I do not want on this bar chart, I will type some other chart let's say as pi. This will give me a pie chart here. Now if I want to select this, if I agree with this answers, I can click on this okay button. So when I click on this okay button, this will convert into a pie chart. I can just drag and drop it here. Not only pie chart, you can decide which visual do you want. So I can go back here, click on Q&A. Here I can talk about let's say profit by segment. segment is a column name in the data set and I want to see in the donut chart. So profit by segment and I can just click on this okay button. This will create a simple donut chart over here. Now let's see some more examples. If I go back and select the option Q&A here I can talk about let's say profit by state. State is a column name. And whenever you have a geographical data remember you should create a map. So now when I say map, it created a map of United States because the data is for United States right now. So I can just click on okay button here. Now this is the profit by state option over here. Now I can just put it down somewhere and put it here. Now similarly if I want again I can just click on this Q&A option and suppose if I only want profit I don't want anything else just profit I can click on okay. This would create a simple cart. We have learned about how to create a card in PowerBI. Again coming back, I can just click on this Q&A. I can again talk about another numerical column. Let's say a sales. Click on okay. This will create as many cards as I want. Now here in this case when I have this cards over here, pie chart, tuner chart and map. I can also create different type of visuals. Let's try one more. I can click on Q&A and here I can write as profit by order date. So remember whenever you have a date data type you should always create a line chart. By default it has created a line chart. I'll just click on okay button here. Now this has created a line chart. Now the best part about this Q&A feature is it understands your simple language and converts into a visual textual whichever format you want. I can click on any of these slice and you see the whole report gets filtered. So in short you're preparing a report a car dashboard in very few seconds. So this is one of the interactive reports/dashboard. Now this is one way of asking questions. You can be more specific and ask them different variety of questions. Let's ask another question here. The question would be give me the top three states by pro. So it will filter and give you the top three states by profit. When I click on okay, what will happen is you're getting three states. But what actually happened was on the right hand side if you see it has filtered the data based on the state. Okay, California, New York and Washington. These are three states. I can go back ask another type of question on this Q&A. This was top three. Suppose if I want the bottom three states by prop. So this will give me the lowest profitable states. So these are three particular states here. Now imagine you're typing and asking questions like in Google or maybe in chat chippity or part they are very simple straightforward questions and it gives you the answer. So if I select this Q&A option and if I want to ask them what is the profit in California okay so it tells me 76.38,000. If you want to verify I can click here I can click on this button data label and you can see 76,000 is a value which is given. So this is for California. But again if you want to be specific for I can ask them for let's say as furniture. Okay. For furniture it is 9.16,000. Now you can be very specific and ask them detail. Okay. So let me revise the question here. How much is the profit made? Let's say if I go back to data view in this column there's a customer. You can ask them about any customer. Let's say the name of the customer is John B. So how much is a profit I've got from this person John B. Let's ask them. So, profit received by John Lee. Okay, even you can remove this part if you don't want it. You can just remove that receive part. So, I'll write as profit by John Lee. It is double28.9. Let's try and test it if it's working or not. So, I'll click on this button here. I'll put here the customer name somewhere customer name and I can select for the profit. So, the name of the person was John Lee. Let me scroll down and search for that person's name John B. So the profit should match. If you see 228.91 if I ask about the John Brier minus 266. So I'll just change the spelling as Brier. Here I'll remove this Lee option T R Y E R. And here let's see it gives me the answers minus 266.55. So this Q&A feature is interesting feature which you can explore all about. Now this is just a start. You can see there are many things which you can explore. If I go back into Q&A here we have a very small settings button. So this will help you to refine your search terms. If I click on settings button you can find the similar words in the field. You can also review questions. You can also suggest questions. You can teach Q&A. Multiple things you can do. If you want to understand better, you can just click on learn more about Q&A. I would suggest you to explore this. When you click on this button, it will take you to the official website of Microsoft where in detail they will try to explain how it can be used. It's a very vast topic and the name of this topic is Q&A. Now it has been a prefix with natural language Q&A. That means you don't need to be a person with knowledge of coding to ask the questions and get the answers. You can easily get the answers by writing a simple English language. The topic is vast and you can just keep on asking questions to get the answer. There are some limitations as well which you can just go through and understand what are the li limitations there. So coming back to our topic the Q&A feature is AI in BI when I say AI is the artificial intelligence in BI which is business intelligence. Now this feature was introduced in the year 2020 long back before chat GP and B has came in the market. So coming back to our topic, I hope you have understood how to use this Q&A feature in PowerBI. And that's all for this video. In this video, we'll learn about the next feature about EI, which is decomposition tree in PowerBI. Now, this is the second video in EI and BI, artificial intelligence in business intelligence. What is the decomposition tree and how do we use it? Let's try to understand. Sometimes if you want to get to the root cause of any problem, we use this kind of decomposition tree. So let's see one example here. This visual is present in the AI visuals and this is the decomposition tree which is available. I can click on this decomposition tree. Now I want to find out the details about high profit and the low profit. So here comes the feature like analyze, explain by. These are the two options very important options. In the analyze, I'll pick up the option as profit. in the explain by I'll pick up here category I'll pick up here region and here I can pick up let's say segment so I want to analyze the profit and I should get the answers by this three reasons okay these are the three problem statement let's say category region and segment now when I say the total profit if you see it's 286,000 that's a value over here that's a total profit now if I want to split this data I can click on this plus sign based on which field I want to understand let's try to talk about the category option. It tells me that the technology is having the highest profit and furniture is having the lowest profit. So technology is having 145,000 and the furniture is having 18,000. Now let's talk about the technology. In technology I want to take down about the region field. So I'll select the region option. Here it tells me the east region, west, south and central. The highest profit received is from east region and the lowest received is from which south region here. And the last one I can click on any plus sign. Let me click on the west plus sign and I can select the option seg. So it tells me that the west region is having the highest profit from the consumer segment and the lowest from the home office segment. Now these are the three different levels I have picked up. I can also add multiple fields and go in the detail part. Now the beautiful part is carefully observe the text which is in the bold it means it has been selected. If you observe here the sum of profit is in bold technology and the west region is in bold. If I select office supplies here it will become as bold. This will give me the details about this particular office supplies here. And if I select east region it's giving the detail about east. So if a person walks in that room and wants to understand what exactly is the visual trying to tell about you can say it's sum of profit for the office supplies and for the east region. On the other hand you can also see on the heading of this particular visual it says the category is filtered for office supplies and the region is given as east. Now here if you do not like this particular visual you can just remove this field. Remove this field and you can remove back. Now again if you want to bring it back you can just click on this plus sign. Now you can select segment. Earlier we have selected category option but this time I'll select segment. It tells me consumer is having the maximum profit, home office the lowest profit. Click on plus select category then click on plus select region. And here it is very flexible to give the answers quickly for any field which you select on the screen. Now if you think this is fine and this has to be presented, you can block this field here. You can lock this level. You can lock all the levels here. So when you lock here, the delete option on the screen will not be visible and you will not be able to change anything. So neither you neither the end user can change. You as a developer can unlock this field again by selecting this option. The moment you unlock you can delete this option here. Now here if I unlock I can delete the lower field the child fields from here from here or this place. Now this was about creation of particular tick composition. Remember this is the AI visual artificial intelligence. Now about the formatting the options about the formatting is on the right hand side here. I can just click on more options. There are plenty of options which are very easy and simple to understand. But something which are very unique to this particular visual are conditional format. If I go back to conditional formatting here I can select this option as sum of profit because there's only one field right now. the data bars. If you see the data bar color right now, it has given the light color and the dark color based on some understanding. However, we can click on this f ofx button to change the colors here. So, in this case, it is already selecting the sum of profit the field it is showing us the gradient format. So, instead of gradient, I think the rules should be much better. So for the rules I can tell them if the value is greater than zero and less than maximum number. So greater than equal to 0 maximum number then the color should be blue which is already there. However I can also select here that the minimum should be empty here. That means it can be anything lowest but it should be less than zero then it will be as a red color. I'm talking about the negative numbers in second line and positive numbers in the first line. When I click on okay, carefully observe within few seconds you can quickly understand that central region is a region which is having lower profit in fact negative profit. It shows in the red color. When you keep on selecting the options you can quickly understand which one is having the negative figure. So I can select something I can select here and quickly get the answers. So this kind of conditional formatting you can bring. This was about the rules. You can talk about the gradient and the field values as well. Now this is one of the conditional format. The other options which are available are the three options here. Here if you see we have the option of density which is default. I can select the option tense. It will compress the visuals. If I say sparse it is spread across. It will occupy the maximum possible space. When you say the dress it will occupy the minimum possible space by trends. The default action which is given as filter and highlight. I can select the option is collapse as well. So collapse and filter these are the options I can pick up from here and it can minimize. You see if I double click it minimizes. If I double click or single click it minimizes. So this is the collapse feature which is given here. Now the other option is filter and highlight. Filter highlight will give you the plus sign on the screen. You can quickly get the answers from here. Now this is what you call the tree settings. The other options we have about the bar options where I can increase or decrease the size of the bars you can see to fit the other visuals as well. I can also define the start and the end point as well. So this is how you can change the layout the positions or the size of those vertical visuals. Now the values option given here if you see the values which are given they are the gray color values. I can change into a black color. I can increase the size which are visible for any users. the decimal values as zero and the display units I can make it as thousands. So this would be much more easy for the end user to read the data. Right? So this was about the EI visual which was the decomposition tree. I hope you have understood and that's all for this video. Welcome back. In this video we'll learn about an interesting feature about PowerBI which is smart narrative. Now this is one of the AI visuals in PowerBI. Let's try to understand whenever you go for a presentation and you have the report or visuals present on your screen it becomes difficult to understand and speak about the data. Now smart is something which will read between the lines. It will try to tell you that visuals that answers even though if they are visually present but still you're not able to decode. Smart derivative is something that kind of visual here. Let's try to understand for time being I'll just create three visuals. The first one I'll create a simple donut chart. Any donut chart you can prepare. So the donut chart would be for the region and sales. I'll just copy paste.trl C and Ctrl P. This region I will replace with the other option given as segment. Ctrl C and Ctrl V. And the third visual I'll convert into a column chart. And this one I will change into let's say as subcategory. So x-axis I can choose here subcategory and this will give me a simple bar chart here. Now when you go for a presentation there's one interesting thing which you have to take care. Let me first change the data labels on the right hand side. The data labels right now are in the millions. Let me convert into thousands decimals as zero and the color I'll change into black. So right now when you go for a presentation usually a person will tell that the mobile phones is having the highest sales, fast is having the lowest sales. Consumer segment is having the highest sales and then the home office segment is having the lowest sales. What is present in the visual anybody can read and tell but what we want to understand is in detail which is not visible which is what we have to decode for the end user. Now this is an interesting feature smart narrative. Creating these three visuals does not come into smart narrative. This is the basic thing what we have learned earlier. Here you have to click outside somewhere on the blank space. Click on this drop-down and here comes the smart narrative. This feature is also present in the insert menu bar which is here which is a smart narrative option. So I can click on smart narrative. It will decode all the answers within few seconds and give me the details. Let's try to understand. It talks about something which says west has the highest sales which is this much amount followed by east, central and south. This is talking about the first visual which is a pie chart. So from here west region is having the highest sales. How much is the west region? It has accounted for 31%age. If you see this is the value 31.58%. It says mobile phone is having the highest sales which is this much higher 11,000 or 10,000 times higher. That's a percentage and which has the lowest value which is this 3,000. So if you compare the mobile phones this value is 10,000 times higher as compared to faster. So this is changing definitely this is giving the answer and it says across all 17 subcategories the values are starting from 3,000 to 330,000. Now whatever you see in the underline this values are dynamic. This will change based on the selection of data on the screen. Suppose if I select the mobile phones over here you see the whole data has been changed. Here I'm getting some other options here some other options and other things. If I change something in the west region the whole report changes. It is trying to give me the answers which will take some hours for me to understand. If I select let's say the consumer segment for consumer segment it says west region at the highest in the consumer. If you ask some person to find out in the consumer which region is having the highest sales it would be difficult because if you see everywhere it looks similar the pie chart looks similar but using this method using the smart native you can quickly get the answers here. So this is the smart nerative option. If you think you want to add some text over here, you can definitely write here the total profit is you can write the total profit. It's a text over here and here in this box I can add a value. So this value I can write as profit. So here when I scroll down it's very small over here. When you scroll down you'll be getting 286,000 and I can click on sale. So this value which is created it's a dynamic value. I can change it. I can also write the total sales is and here I can just go to add value. I can talk about sales and here when I scroll down I'll be getting sales as 2.29 and click on save button. So I can create more complex uh dynamic inputs which are required here and still get the answers. Now this will definitely change based on selected value, selected bar or slice of the pie chart. So smart narrative is something which will speak what a end user will not be able to decode even though the visuals are present. Okay, this is one of the EI visual which was recently launched uh in the PowerBI itself. So I hope you have understood about the smart narrative in PowerBI and that's all for this video. In this video we'll see one more AI feature in PowerBI which is explain the increase and decrease. Now uh this is not present anywhere in this visual. It's kind of a hidden feature in PowerBI. But let's try to understand very very important. There are many people in the company which spends hours together to find out the root cause or the detail about why this thing happened. An example I want to find out why did the profit decrease? What was the reason behind that? Why did the sales decrease? Let me show you an example. Suppose on the right hand side of this data I have this option given as profit. There might be some places where the profit has declined. It can be because of the customer didn't come because of the city didn't perform or the category was not sold. It can be because of segment because of subcategory anything. The reason can be anything but how do I get the answers? That's a beautiful part right now in PowerBI. Let's see. First thing to get the understanding of this particular AI feature. I'll create a simple column chart for order date and sales. This created a column chart. I will add here data labels as well. Now in this data label when I get it, I'll just click on this expand all down one level in hierarchy. So this will create simple column chart which shows me the years as well as the quarters. And here the data label which are given it's in millions. Let's try to convert into thousands. On the right hand side I'll go to data label. Convert this color as black color value decimals as zero and display units as thousands. So I've got this data label in a proper manner. Now coming to this point over here. Now let's say assume the company right now is in 2020 Q4. Okay, which is 236,000. Now the profit has came down or the sales have come down to 123,000. What could be the reason? Now the company wants to find out the root cause so that they can take an action and then they can increase the seats. Similar trend you have observed many places. Now 182,000 the value has declined to 93,000. What could be the reason? The reason can be any field on the right hand side which is difficult to decode. To get this answers, people spend 3 to four hours in the organization desperately to get the answers here. Now here, let's try to get the answers in less than 60 seconds. First thing from 236,000 the sales have come down to 123,000. What could be the reason? You don't have to ask anyone else. You have to just ask this 123,000 which is 2021 Q1. I can right click on this particular field and I can tell them analyze explain the decrease. Don't ask anyone else. You can just directly ask them. Explain the decrease and in less than 60 seconds it will give you all the answers. The first answer it says how much percentage the value has decreased. It is 47.84%. And if you scroll down it has given already many visuals to tell you why the sales have decreased. Let's try to understand the first visual. If you observe here from 236,000 the value has decreased to 123,000 that is 2020 Q4 to 2021 Q1 and the main reason there are many regions which the profit has decreased but the main region and the reason is east region it says minus 48,000 this was the reason second could be central and third and four this was about the region not only this it also gives us in the writing format in the English format East and central accounted for majority of decrease. East and central accounted for the majority of decrease among the region. The relative contributions made by the east and west change the most. Okay. So it gives me the answers here. Let's try to understand one. From 236,000 it came down to 123,000. And what could be the reason? Corporate segment was the reason. It is also written on the top it says corporate accounted for the majority of decrease among the segment and the relative is the corporate most. So second reason it has given me and if I go somewhere on the top it will give me all the answers here. And third one which is says which category was accounted for majority of degrees. It says the furniture again - 57,000k. All these answers we have got in less than 60 seconds. Usually people take 3 to 4 hours minimum to get the answers. Now here coming back to the next example. Now your company did a sales of 144,000 in Q3 and there was a spike in the sales which has come down to come to 236,000. Now the company wants to give the profit and you know within the team it has to distribute the profit in terms of incentives or goodies or something else but they are not sure this 236,000 increase in sales was done because of which corporate client because of which region and so on. So if they want to get the answers they can select this 236,000 right click on this place and then analyze explain the increase. It can quickly explain the increase in less than 60 seconds. First answer by how much percentage the sales have increased 64.20%age and all these visuals are created. The first answer let's try to see corporate segment was something because of which the sales have increased by 50% or I can say 50,000. So the management can take a decision that the goodies or incentives or awards should go to this particular corporate segment. Similarly, the second example it says the standard class this is the type of shipping mode standard class has got 65,000 which is a huge number because of which the sales have increased. And let's say one more and the last right and here it says based on which subcategory we have got the highest sales. It says others all of them have contributed as 30,000 but major one if you see the binders which is 16,000 then 15 then 12 and then so on. So the company can decide about this particular subcategory it has to be better packaging can be done something better and we have got the answer. Now we have seen 123 to the previous one. We have seen 236 with the previous one. But what if I want to do Q and Q analysis quarter on quarter analysis? Can we do that? I can select any quarter of 2018 and any quarter of let's say 2021. Okay, this both are different quarters. They don't have relation between them. But still I want to understand the value 144,000 to 280,000 how it has come. So I can select both of these bars. Right click analyze and I can say explain the increase. This will explain with the increase and you see 2018 quarter 3 with 2021 quarter 4 how much percentage 94.98%. And if you scroll down you'll get the same story what we have got there. Now the beautiful part right now is suppose if you want to add this visual if you think this visual is important and when you show this to a management they'll be happy and you can definitely include this visual. You can click on this plus sign and this visual would be added. You have to add this visual after 15 days. After 15 days, you have to click on this plus button. For time being, I'll just click it right away. When you do this, what will happen is this particular visual what we selected will be added over here in this place. And now when you go for a presentation, you can show this particular waterfall chart which is simple and easily available. Now let's talk about one more example here. Suppose if I right click on this place and I say analyze, explain the increase. Here when I scroll down this is a boring chart which is a waterfall chart but in future you may require some fancy visuals like this is the second visual this is the third visual and fourth they all have the same meaning but the way of presentation is different. This is the stack bar chart ribbon chart and the scattered plot. You can also bring this kind of charts by selecting you can click on this add button after 15 days and this particular visual would be added at the back end. Now this is something very interesting. You can keep on adding getting the answers. Now why do we add after 15 days? When you add it immediately and show it to your manager the manager will give you the another work. He will say very good job. He'll appreciate for next 2 3 minutes and he'll give you the another work for you. So when you add after 15 days you can tell them sir I've taken lot of effort to get the answers. It took me almost 15 days and night to get the answers on a lighter note but you should give them immediately depending upon your relation with your manager. And now coming to the point over here. Now this feature right click analyze explain the increase and decrease is not there anywhere on this visuals. You have to yourself understand decode and get the answers quickly. Now this works better when you have the x-axis as a date data type. When you have a text data type it may or may not work properly. It will give you some different answers. So remember whenever you have a date data type in the x-axis this will give you the right answers. So I hope you have understood how to work on this explain the increase and decrease feature in PowerBI. And that's all for this video. Welcome back. In this video, we'll learn about the next AI feature in PowerBI which is how to get the quick insights in PowerBI. Now even if you have not created any visual in PowerBI and just loaded the data just clean the data by one button click you'll be getting approximately 20 visuals by default. That is the beautiful part of PowerBI. Let's try to understand here on the right hand side I have just loaded the data. Even if I have no pages over here still this feature will work. What I need to do I just need to publish this particular workbook in the online version. To publish you need a corporate email account only then you can work on that. Right now I'll just click on save button. Then I can click on publish my workspace. Click on select. It will take few seconds depending upon the size of data depending upon your internet connection. In less than 60 seconds this will publish the report online. Now if you want to open the report you can click on this open this particular visual in the PowerBS service or you can just click on this get quick insights. What is this name? This is the name of the workbook right now EI in BI. But currently we are talking about get quick insights. Remember this feature will work even if you don't have the data. Even if you don't have the visuals in all those 10 pages, you just need to have a data. It should be clean data. And then if I click on get quick insights, what will happen is it will try to create more than 20 plus visuals for you. What it will do is it will create more than 20 visuals with the detailed understanding. So if you see all these visuals are created. I have created none of them right now. So all these visuals are created and they are ready to use. You can just pin them on the dashboard directly. So you see all these visuals. So let's try to understand the first visual on the top. So the first one says it's the postal code by category. This does not make any sense. Though postal code is a numerical column but it is a geographical location. This cannot be used. It says profit by city. Yes, you can use this particular visual. It says that New York City is having the maximum profit. You see, it is also written for our understanding. New York City has the noticeably more profit here. If you see the profit by product ID, it says this particular product ID has having the lowest profit. Sales by state, then sales and row ID by order. It has tried to do all the permutation and combinations to get the different visuals and answer. We may not use all of them. We may use some of this particular visuals. So let's say if you like this particular chart and you want to use it in this dashboard, you cannot use in the report. You have to use in the dashboard. Now what is a report in dashboard? Difference between them it is covered in the coming section. So for time being you understand this visuals wherever you get a pin icon this icons will be pinned and the final destination would be the dashboards not the report. So when I click on this button over here, pin visual, it says existing or new dashboard. I'll say new dashboard. New dashboard. I'll write here as PowerBI training. And here I can just click on pin. Now this particular visual would be pinned to the dashboard. I can go to the dashboard directly from here. But before I go, let's add one more visual. I can just click on pin visual. Now I'll select existing dashboard. From here I'll select the option as PowerBI dashboard. The name of this report or the dashboard was powerp dash right the name of this one is powerp training. I can select this existing dashboard. Click on pin. And last one I can again select this particular donut chart existing dashboard. And here I have to select for powerp training and click on pin button. Now I can go to the dashboard directly from here. When I click on this you see all those three visuals are added onto this particular dashboard. Here these are three uh visuals which are created by EI. Now in this dashboard if you think one chart is missing you can definitely type over here which is ask a question. I can click on this box. This is a Q&A feature in the dashboard. We have seen Q&A in the desktop version but in the dashboard also you'll be getting this Q&A feature. Now you can come back if you want. Click on exit Q&A. I can click on this box. It will open here. You can also refresh the browser if the data is not loaded up to date. Now again if I click on ask a question it is preparing for a Q&A and here I can ask them let's say uh sales by subcategory infunnel sales by subcategory infunnel and here on the right hand side I can click on pin visual existing dashboard and the dashboard name is power train here I can select click on pin button and I can go back to the dashboard when And I see here you see the fourth visual which is a funnel chart which was missing. I never created in the desktop version. So still even if you don't create it during the runtime you can get the answers. Let's try one more. If I click on ask Q&A I can write here as top three states by problem. All right. And then I can just click on pin visual existing dashboard. And here I can select as powerbi trading. Click on pin. Now if you want to come back you can click on exit Q. Now one thing you have to note which is very important when you click on this publish button and save. I'm just again repeating the same steps. Replace and then click on get quick insights. Once you do that then you will be getting multiple visuals on this platform. Now there is a less chances that you might trust on this particular uh feature. The reason I'll be telling you this takes a subset of the data. It doesn't take the whole data and gives us the answer. It's written on the top. So the subset of data was analyzed and following insights were found that now how much data it has taken. Has it taken 1 GB data or 10,000 data or 1 billion records. There's no clarity about that. Even if you click on this learn more feature, learn more button, it will take you to official website. It doesn't tell us how much data it is actually capturing. It creates visuals. No doubt about that. But imagine if you have billions and billions of record. Now you cannot be pretty sure about it because a subset of your data was analyzed. The previous examples what we have seen different AI visuals there the whole data set has been picked up. You don't take a partial data. So in this case right now a subset of data. Now I have personally checked it for 1 billion records. It works fine. But still based on this line it becomes difficult to give you the right answers. So 1 billion records it works fines and gives us the answer properly. Perfect. So I hope you have understood this particular AI feature in PowerBI which is get quick insights in PowerBI. And that's all for this video. \ No newline at end of file diff --git a/graphrag-ollama-config/openai_config.py b/graphrag-ollama-config/openai_config.py new file mode 100644 index 0000000..3727078 --- /dev/null +++ b/graphrag-ollama-config/openai_config.py @@ -0,0 +1,170 @@ +""" +OpenAI-Compatible API Configuration Module + +This module provides configuration utilities for connecting to OpenAI, Azure OpenAI, +and any OpenAI-compatible API endpoints (e.g., Ollama, LM Studio, Groq, Together AI, etc.) + +Environment Variables: + GRAPHRAG_API_KEY: API key for the LLM provider + GRAPHRAG_API_TYPE: API type ('openai' or 'azure'), defaults to 'openai' + GRAPHRAG_LLM_MODEL: Model name for chat completions + GRAPHRAG_LLM_API_BASE: Base URL for LLM API + GRAPHRAG_EMBEDDING_MODEL: Model name for embeddings + GRAPHRAG_EMBEDDING_API_BASE: Base URL for embedding API + GRAPHRAG_MAX_RETRIES: Maximum retry attempts (default: 10) + +Azure-specific: + GRAPHRAG_API_VERSION: Azure API version + GRAPHRAG_DEPLOYMENT_NAME: Azure deployment name for LLM + GRAPHRAG_EMBEDDING_DEPLOYMENT_NAME: Azure deployment name for embeddings + +Optional: + GRAPHRAG_ORGANIZATION: OpenAI organization ID + GRAPHRAG_EMBEDDING_API_KEY: Separate API key for embeddings (if different) + GRAPHRAG_EMBEDDING_API_TYPE: Separate API type for embeddings +""" + +import os +from graphrag.query.llm.oai.typing import OpenaiApiType + + +def get_api_type() -> OpenaiApiType: + """ + Determine the API type based on environment variables. + Supports: OpenAI, Azure OpenAI, and any OpenAI-compatible API. + + Set GRAPHRAG_API_TYPE to 'azure' for Azure OpenAI, otherwise defaults to OpenAI-compatible. + + Returns: + OpenaiApiType: The API type enum value + """ + api_type_str = os.environ.get("GRAPHRAG_API_TYPE", "openai").lower() + if api_type_str == "azure": + return OpenaiApiType.AzureOpenAI + return OpenaiApiType.OpenAI + + +def get_llm_config() -> dict: + """ + Get LLM configuration from environment variables. + Supports OpenAI, Azure OpenAI, and OpenAI-compatible APIs. + + Returns: + dict: Configuration dictionary with keys: + - api_key: API key for authentication + - api_base: Base URL for the API + - model: Model name to use + - api_type: OpenaiApiType enum value + - max_retries: Maximum number of retries + - api_version: (Azure only) API version + - deployment_name: (Azure only) Deployment name + - organization: (Optional) Organization ID + """ + api_type = get_api_type() + + config = { + "api_key": os.environ["GRAPHRAG_API_KEY"], + "api_base": os.environ.get("GRAPHRAG_LLM_API_BASE", "https://api.openai.com/v1"), + "model": os.environ.get("GRAPHRAG_LLM_MODEL", "gpt-4-turbo-preview"), + "api_type": api_type, + "max_retries": int(os.environ.get("GRAPHRAG_MAX_RETRIES", "10")), + } + + # Azure-specific configuration + if api_type == OpenaiApiType.AzureOpenAI: + config["api_version"] = os.environ.get("GRAPHRAG_API_VERSION", "2024-02-15-preview") + config["deployment_name"] = os.environ.get("GRAPHRAG_DEPLOYMENT_NAME", config["model"]) + + # Optional: Organization ID for OpenAI + org_id = os.environ.get("GRAPHRAG_ORGANIZATION") + if org_id: + config["organization"] = org_id + + return config + + +def get_embedding_config() -> dict: + """ + Get embedding configuration from environment variables. + Supports OpenAI, Azure OpenAI, and OpenAI-compatible APIs. + + Allows separate configuration for embeddings (useful for hybrid setups, + e.g., using Groq for LLM and Ollama for embeddings). + + Returns: + dict: Configuration dictionary with keys: + - api_key: API key for authentication + - api_base: Base URL for the API + - model: Model name to use + - api_type: OpenaiApiType enum value + - max_retries: Maximum number of retries + - deployment_name: Deployment name (required for Azure, model name for others) + - api_version: (Azure only) API version + """ + api_type = get_api_type() + + # Allow separate API type for embeddings (useful for hybrid setups) + embedding_api_type_str = os.environ.get("GRAPHRAG_EMBEDDING_API_TYPE", "").lower() + if embedding_api_type_str == "azure": + embedding_api_type = OpenaiApiType.AzureOpenAI + elif embedding_api_type_str == "openai": + embedding_api_type = OpenaiApiType.OpenAI + else: + embedding_api_type = api_type # Default to same as LLM + + config = { + "api_key": os.environ.get("GRAPHRAG_EMBEDDING_API_KEY", os.environ["GRAPHRAG_API_KEY"]), + "api_base": os.environ.get("GRAPHRAG_EMBEDDING_API_BASE", "https://api.openai.com/v1"), + "model": os.environ.get("GRAPHRAG_EMBEDDING_MODEL", "text-embedding-3-small"), + "api_type": embedding_api_type, + "max_retries": int(os.environ.get("GRAPHRAG_MAX_RETRIES", "10")), + } + + # Azure-specific configuration + if embedding_api_type == OpenaiApiType.AzureOpenAI: + config["api_version"] = os.environ.get( + "GRAPHRAG_EMBEDDING_API_VERSION", + os.environ.get("GRAPHRAG_API_VERSION", "2024-02-15-preview") + ) + config["deployment_name"] = os.environ.get("GRAPHRAG_EMBEDDING_DEPLOYMENT_NAME", config["model"]) + else: + config["deployment_name"] = config["model"] + + return config + + +def validate_config() -> tuple[bool, list[str]]: + """ + Validate that required environment variables are set. + + Returns: + tuple: (is_valid, list of error messages) + """ + errors = [] + + if not os.environ.get("GRAPHRAG_API_KEY"): + errors.append("GRAPHRAG_API_KEY is required") + + api_type = os.environ.get("GRAPHRAG_API_TYPE", "openai").lower() + if api_type == "azure": + if not os.environ.get("GRAPHRAG_DEPLOYMENT_NAME"): + errors.append("GRAPHRAG_DEPLOYMENT_NAME is required for Azure OpenAI") + + return len(errors) == 0, errors + + +def print_config_summary(): + """Print a summary of the current configuration (for debugging).""" + api_type = get_api_type() + llm_config = get_llm_config() + embedding_config = get_embedding_config() + + print("=" * 50) + print("GraphRAG Configuration Summary") + print("=" * 50) + print(f"API Type: {api_type.value}") + print(f"LLM Model: {llm_config['model']}") + print(f"LLM API Base: {llm_config['api_base']}") + print(f"Embedding Model: {embedding_config['model']}") + print(f"Embedding API Base: {embedding_config['api_base']}") + print("=" * 50) diff --git a/graphrag-ollama-config/test_chat.py b/graphrag-ollama-config/test_chat.py new file mode 100644 index 0000000..059641d --- /dev/null +++ b/graphrag-ollama-config/test_chat.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +"""Simple test script for the Gradio chat interface.""" +import requests +import json + +def test_gradio_api(): + """Test the Gradio API endpoint.""" + base_url = "http://127.0.0.1:7861" + + # First, check if the server is responding + try: + response = requests.get(f"{base_url}/") + print(f"✓ Server is running (status: {response.status_code})") + except requests.exceptions.ConnectionError: + print("✗ Server is not running. Please start the app first with: python app.py") + return + + # Test the API info endpoint + try: + response = requests.get(f"{base_url}/info") + if response.status_code == 200: + print(f"✓ API info endpoint is working") + else: + print(f"○ API info endpoint returned: {response.status_code}") + except Exception as e: + print(f"○ API info: {e}") + + # Test the chat function via Gradio's API + try: + # Gradio uses a specific API format + api_url = f"{base_url}/api/predict" + + # Check what endpoints are available + config_response = requests.get(f"{base_url}/config") + if config_response.status_code == 200: + print(f"✓ Config endpoint is available") + config = config_response.json() + print(f" App title: {config.get('title', 'N/A')}") + print(f" Components: {len(config.get('components', []))} found") + + print("\n--- Chat Test ---") + print("Note: The chat functionality requires indexed GraphRAG data.") + print("Since the output folder is empty, search queries will fail gracefully.") + print("To fully test, you need to run the indexing pipeline first:") + print(" python -m graphrag.index --root .") + + except Exception as e: + print(f"Error testing API: {e}") + +if __name__ == "__main__": + test_gradio_api() diff --git a/graphrag-ollama-config/test_reasoning_search.py b/graphrag-ollama-config/test_reasoning_search.py new file mode 100644 index 0000000..874caa7 --- /dev/null +++ b/graphrag-ollama-config/test_reasoning_search.py @@ -0,0 +1,145 @@ +""" +Test script for Reasoning-Based Search +Run from the graphrag-ollama-config directory +""" +import asyncio +import os +import sys + +# Add current directory to path +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +from reasoning_search import ( + enhanced_search, + reasoning_local_search, + reasoning_global_search, + hybrid_reasoning_search, + ReasoningSearchEngine +) + +async def test_reasoning_search(): + """Test the reasoning-based search functionality""" + + script_dir = os.path.dirname(os.path.abspath(__file__)) + input_dir = os.path.join(script_dir, "output", "artifacts") + + if not os.path.exists(input_dir): + print(f"❌ Input directory not found: {input_dir}") + print("Make sure you have run GraphRAG indexing first.") + return + + print("=" * 70) + print("🧪 Testing PageIndex-Inspired Reasoning Search for VeritasGraph") + print("=" * 70) + + # Test queries + test_queries = [ + "What are the main eligibility criteria for student visas?", + "Compare USA F-1 and UK Tier 4 student visa requirements", + ] + + for query in test_queries: + print(f"\n{'='*70}") + print(f"📝 Query: {query}") + print("=" * 70) + + try: + # Test enhanced_search with auto strategy + print("\n🎯 Testing Reasoning Search (auto strategy)...") + result = await enhanced_search( + query=query, + input_dir=input_dir, + query_type="auto", + community_level=2, + temperature=0.3 + ) + + print(f"\n✅ Response received!") + print(f"📊 Confidence: {result.get('confidence', 'N/A'):.2%}") + print(f"✓ Verified: {result.get('verified', 'N/A')}") + + # Print reasoning trace summary + if 'reasoning_trace' in result: + print(f"\n🔍 Reasoning Steps ({len(result['reasoning_trace'])}):") + for i, step in enumerate(result['reasoning_trace'], 1): + print(f" {i}. {step['step']}: {step.get('confidence', 0):.2%} confidence") + + # Print sources + if 'sources' in result and result['sources']: + print(f"\n📚 Sources ({len(result['sources'])}):") + for src in result['sources'][:3]: + print(f" - [{src['type']}] {src['name']}") + + # Print truncated response + response = result.get('response', '') + print(f"\n📄 Answer Preview:") + print("-" * 50) + if len(response) > 500: + print(response[:500] + "...") + else: + print(response) + print("-" * 50) + + except Exception as e: + print(f"❌ Error: {e}") + import traceback + traceback.print_exc() + + print("\n" + "=" * 70) + print("✅ Reasoning Search Test Complete!") + print("=" * 70) + +async def compare_search_methods(): + """Compare standard vs reasoning search""" + + script_dir = os.path.dirname(os.path.abspath(__file__)) + input_dir = os.path.join(script_dir, "output", "artifacts") + + if not os.path.exists(input_dir): + print(f"❌ Input directory not found: {input_dir}") + return + + query = "What financial requirements exist for student visas?" + + print("=" * 70) + print("📊 Comparing Search Methods") + print("=" * 70) + print(f"\nQuery: {query}\n") + + # Test reasoning search + print("🎯 Method 1: Reasoning Search (PageIndex-style)") + print("-" * 50) + try: + result = await enhanced_search(query, input_dir, "auto", 2, 0.3) + print(f"Confidence: {result.get('confidence', 0):.2%}") + print(f"Verified: {result.get('verified', False)}") + print(f"Sources: {len(result.get('sources', []))}") + print(f"Response length: {len(result.get('response', ''))} chars") + except Exception as e: + print(f"Error: {e}") + + # Test hybrid search + print("\n🔄 Method 2: Hybrid Search") + print("-" * 50) + try: + result = await hybrid_reasoning_search(query, input_dir, 2, 0.3) + print(f"Confidence: {result.get('confidence', 0):.2%}") + print(f"Strategy: {result.get('search_strategy', 'N/A')}") + print(f"Response length: {len(result.get('response', ''))} chars") + except Exception as e: + print(f"Error: {e}") + + print("\n" + "=" * 70) + print("Comparison complete!") + +if __name__ == "__main__": + print("\n" + "=" * 70) + print("VeritasGraph Reasoning Search Test Suite") + print("Based on PageIndex's 98.7% accuracy approach") + print("=" * 70 + "\n") + + # Run tests + asyncio.run(test_reasoning_search()) + + print("\n\n") + asyncio.run(compare_search_methods()) diff --git a/indexing_output.log b/indexing_output.log deleted file mode 100644 index 6f7cfa2..0000000 --- a/indexing_output.log +++ /dev/null @@ -1,7 +0,0 @@ -Traceback (most recent call last): - File "", line 189, in _run_module_as_main - File "", line 148, in _get_module_details - File "", line 112, in _get_module_details - File "/home/sijo/VeritasGraph/graphrag-ollama-config/graphrag-ollama/graphrag/index/__init__.py", line 6, in - from .cache import PipelineCache -ModuleNotFoundError: No module named 'graphrag.index.cache' diff --git a/pyproject.toml b/pyproject.toml deleted file mode 100644 index 0fa790c..0000000 --- a/pyproject.toml +++ /dev/null @@ -1,165 +0,0 @@ -[build-system] -requires = ["hatchling>=1.18.0"] -build-backend = "hatchling.build" - -[project] -name = "veritasgraph" -version = "0.3.0" -description = "Enterprise-Grade Graph RAG Framework - Don't Chunk. Graph. Vision-Native Document Analysis with Hierarchical Tree Support" -readme = "README.md" -license = {text = "MIT"} -requires-python = ">=3.10" -authors = [ - { name = "Bibin Prathap", email = "bibinprathap@gmail.com" } -] -maintainers = [ - { name = "VeritasGraph Team" } -] - -keywords = [ - "rag", - "graphrag", - "vision", - "llm", - "document-analysis", - "pdf", - "multimodal", - "knowledge-graph", - "ollama", - "enterprise-ai", - "retrieval-augmented-generation", - "graph-database", - "hierarchical-tree", - "table-of-contents", - "ai-reasoning" -] -classifiers = [ - "Development Status :: 4 - Beta", - "Intended Audience :: Developers", - "Intended Audience :: Science/Research", - "Intended Audience :: Information Technology", - "License :: OSI Approved :: MIT License", - "Operating System :: OS Independent", - "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.10", - "Programming Language :: Python :: 3.11", - "Programming Language :: Python :: 3.12", - "Programming Language :: Python :: 3.13", - "Topic :: Scientific/Engineering :: Artificial Intelligence", - "Topic :: Text Processing :: General", - "Topic :: Software Development :: Libraries :: Python Modules", - "Typing :: Typed", -] -dependencies = [ - "ollama>=0.2.0", - "pillow>=10.0.0", - "pdf2image>=1.16.0", - "networkx>=3.0", - "numpy>=1.24.0", - "matplotlib>=3.7.0", - "python-dotenv>=1.0.0", -] - -[project.optional-dependencies] -dev = [ - "pytest>=7.0.0", - "pytest-asyncio>=0.21.0", - "pytest-cov>=4.0.0", - "black>=23.0.0", - "ruff>=0.1.0", - "mypy>=1.0.0", - "pre-commit>=3.0.0", -] -graphrag = [ - "graphrag>=0.3.0", - "tiktoken>=0.5.0", -] -web = [ - "gradio>=4.0.0", - "pyvis>=0.3.0", - "pandas>=2.0.0", -] -ingest = [ - "youtube-transcript-api>=0.6.0", - "trafilatura>=1.6.0", - "yt-dlp>=2024.1.0", -] -all = [ - "veritasgraph[dev,graphrag,web,ingest]", -] - -[project.scripts] -veritasgraph = "veritasgraph.cli:main" -vg = "veritasgraph.cli:main" - -[project.urls] -Homepage = "https://github.com/bibinprathap/VeritasGraph" -Documentation = "https://bibinprathap.github.io/VeritasGraph/" -Repository = "https://github.com/bibinprathap/VeritasGraph" -Issues = "https://github.com/bibinprathap/VeritasGraph/issues" -Changelog = "https://github.com/bibinprathap/VeritasGraph/releases" - -[tool.hatch.version] -path = "veritasgraph/__init__.py" - -[tool.hatch.build.targets.sdist] -include = [ - "/veritasgraph", - "/README.md", - "/LICENSE", - "/pyproject.toml", -] -exclude = [ - "/.git", - "/.github", - "/.venv", - "/.vscode", - "/docker", - "/docs", - "/finetune", - "/graphrag-ollama-config", - "/scripts", - "/output", - "*.pyc", - "__pycache__", - "*.egg-info", -] - -[tool.hatch.build.targets.wheel] -packages = ["veritasgraph"] - -[tool.black] -line-length = 100 -target-version = ["py310", "py311", "py312"] - -[tool.ruff] -line-length = 100 -target-version = "py310" -select = ["E", "F", "W", "I", "UP", "B", "SIM"] -ignore = ["E501", "B008"] - -[tool.ruff.isort] -known-first-party = ["veritasgraph"] - -[tool.mypy] -python_version = "3.10" -warn_return_any = true -warn_unused_configs = true -ignore_missing_imports = true - -[tool.pytest.ini_options] -testpaths = ["tests"] -asyncio_mode = "auto" -addopts = "-v --tb=short" - -[tool.coverage.run] -source = ["veritasgraph"] -omit = ["tests/*"] - -[tool.coverage.report] -exclude_lines = [ - "pragma: no cover", - "if TYPE_CHECKING:", - "raise NotImplementedError", -] - diff --git a/scripts/start-with-pages-sync.sh b/scripts/start-with-pages-sync.sh new file mode 100755 index 0000000..ad61550 --- /dev/null +++ b/scripts/start-with-pages-sync.sh @@ -0,0 +1,224 @@ +#!/bin/bash +# VeritasGraph - Auto-update GitHub Pages redirect with Cloudflare Tunnel URL +# This script starts the app, creates a tunnel, and updates the demo redirect page +# The README stays unchanged - only docs/demo/index.html is updated + +set -e + +# ============================================================================= +# CONFIGURATION - UPDATE THESE VALUES +# ============================================================================= +# Load GITHUB_TOKEN from .env file if not already set in environment +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +ENV_FILE="$SCRIPT_DIR/../.env" +if [ -f "$ENV_FILE" ]; then + export $(grep -E '^GITHUB_TOKEN=' "$ENV_FILE" | xargs) +fi + +if [ -z "$GITHUB_TOKEN" ]; then + echo "ERROR: GITHUB_TOKEN is not set. Add it to .env file: GITHUB_TOKEN=your_token" + exit 1 +fi + +GITHUB_REPO="bibinprathap/VeritasGraph" +GITHUB_BRANCH="restored-main" +REDIRECT_FILE_PATH="docs/demo/index.html" + +PROJECT_DIR="/home/sijo/VeritasGraph/graphrag-ollama-config" +PYTHON_PATH="/home/sijo/VeritasGraph/.venv/bin/python" +LOG_FILE="/home/sijo/VeritasGraph/veritasgraph.log" + +# ============================================================================= +# DO NOT EDIT BELOW THIS LINE +# ============================================================================= + +RED='\033[0;31m' +GREEN='\033[0;32m' +YELLOW='\033[1;33m' +CYAN='\033[0;36m' +NC='\033[0m' + +log() { + echo -e "${GREEN}[$(date '+%Y-%m-%d %H:%M:%S')]${NC} $1" | tee -a "$LOG_FILE" +} + +error() { + echo -e "${RED}[$(date '+%Y-%m-%d %H:%M:%S')] ERROR:${NC} $1" | tee -a "$LOG_FILE" +} + +# Function to update GitHub Pages redirect (uses git push - works with fine-grained PATs) +update_github_redirect() { + local DEMO_URL=$1 + local NOW + NOW=$(date -u '+%Y-%m-%d %H:%M UTC') + + log "${YELLOW}📝 Updating GitHub Pages redirect via git...${NC}" + + # Write the updated redirect HTML directly into the repo's docs/demo/index.html + cat > "$SCRIPT_DIR/../$REDIRECT_FILE_PATH" << HTMLEOF + + + + + + + + VeritasGraph Live Demo - Redirecting... + + + +
+

🔍 VeritasGraph Demo

+
+

Redirecting to live demo...

+ Click here if not redirected +
+ Status: Server is online
+ Last updated: ${NOW} +
+
+ + + +HTMLEOF + + # Commit and push via git using the token + cd "$SCRIPT_DIR/.." + REMOTE_URL="https://${GITHUB_TOKEN}@github.com/${GITHUB_REPO}.git" + + # Configure git user identity if not already set (required for commits in automated environments) + if ! git config user.email > /dev/null 2>&1; then + git config user.email "veritasgraph@localhost" + git config user.name "VeritasGraph Auto-Updater" + fi + + # Fetch and checkout the target branch + git fetch origin "$GITHUB_BRANCH" 2>/dev/null || true + CURRENT_BRANCH=$(git rev-parse --abbrev-ref HEAD) + if [ "$CURRENT_BRANCH" != "$GITHUB_BRANCH" ]; then + git checkout "$GITHUB_BRANCH" 2>/dev/null || git checkout -B "$GITHUB_BRANCH" "origin/$GITHUB_BRANCH" + fi + # Pull latest changes from remote + git pull origin "$GITHUB_BRANCH" --rebase 2>/dev/null || true + + git add "$REDIRECT_FILE_PATH" + if git diff --cached --quiet; then + log "No change in redirect URL, skipping commit." + return 0 + fi + + git commit -m "🔄 Update demo redirect: ${DEMO_URL}" + git push "$REMOTE_URL" "$GITHUB_BRANCH" + + if [ $? -eq 0 ]; then + log "${GREEN}✅ GitHub Pages redirect updated!${NC}" + log "${CYAN}📍 Stable URL: https://bibinprathap.github.io/VeritasGraph/demo/${NC}" + log "${CYAN}📍 Current tunnel: ${DEMO_URL}${NC}" + return 0 + else + error "Failed to push redirect update." + return 1 + fi +} + +# Function to extract Cloudflare URL from output +extract_cf_url() { + grep -oP 'https://[a-z0-9-]+\.trycloudflare\.com' | head -1 +} + +# Main function +main() { + log "============================================" + log "🚀 VeritasGraph - Starting with GitHub Pages Sync" + log "============================================" + + # Start the Gradio app in background + log "Starting VeritasGraph app..." + cd "$PROJECT_DIR" + $PYTHON_PATH app.py & + APP_PID=$! + + # Wait for app to start + sleep 5 + + # Check if app is running + if ! kill -0 $APP_PID 2>/dev/null; then + error "App failed to start!" + exit 1 + fi + + log "✅ App started on port 7860" + + # Handle shutdown + trap "log 'Shutting down...'; kill $APP_PID 2>/dev/null; exit 0" SIGINT SIGTERM + + # Start Cloudflare tunnel and monitor output in the current shell (no subshell) + # Using process substitution < <(...) keeps the while loop in the current shell, + # so update_github_redirect can log errors and git operations are visible. + log "Starting Cloudflare tunnel..." + URL_FOUND=false + while IFS= read -r line; do + # Strip ANSI escape codes that cloudflared injects (breaks grep/regex) + clean_line=$(printf '%s' "$line" | sed 's/\x1b\[[0-9;]*[mGKHFJl]//g') + echo "$clean_line" | tee -a "$LOG_FILE" + + # Extract the tunnel URL when it first appears; skip if already found + if [ "$URL_FOUND" = "false" ] && echo "$clean_line" | grep -q "trycloudflare.com"; then + TUNNEL_URL=$(echo "$clean_line" | grep -oP 'https://[a-zA-Z0-9-]+\.trycloudflare\.com' | head -1) + if [ -n "$TUNNEL_URL" ]; then + URL_FOUND=true + log "✅ Tunnel URL: $TUNNEL_URL" + + # Update GitHub Pages redirect + if update_github_redirect "$TUNNEL_URL"; then + log "🌐 Server is running. Press Ctrl+C to stop." + else + error "GitHub Pages update failed — check token permissions and git config" + fi + fi + fi + done < <(cloudflared tunnel --url http://localhost:7860 2>&1) + + # Tunnel exited — stop the app too + log "Cloudflare tunnel exited. Stopping app..." + kill $APP_PID 2>/dev/null || true +} + +main "$@" diff --git a/veritasgraph/__init__.py b/veritasgraph/__init__.py deleted file mode 100644 index d668240..0000000 --- a/veritasgraph/__init__.py +++ /dev/null @@ -1,133 +0,0 @@ -""" -VeritasGraph - Enterprise-Grade Graph RAG Framework -==================================================== - -Vision-Native RAG with Knowledge Graph Integration for Secure, On-Premise AI. - -"Don't Chunk. Graph." - Document-Centric RAG: -- Treats whole pages/sections as single nodes (not arbitrary 500-token chunks) -- Preserves document structure and visual context -- Better for RAG with tables, charts, and rich formatting - -Hierarchical Tree Support: -- Combines PageIndex's TOC-based "human-like retrieval" -- With the flexibility of graph-based semantic search -- "The Power of PageIndex's Tree + The Flexibility of a Graph" - -Features: -- Vision-based document processing (no OCR needed) -- Tables and charts extraction using multimodal LLMs -- Hierarchical tree structure extraction (TOC-style navigation) -- Parent-Child section relationships in the graph -- Knowledge graph construction from visual content -- Hybrid RAG combining text and visual understanding -- Integration with GraphRAG for enterprise deployments -- Full source attribution for verifiable AI - -Installation: - pip install veritasgraph - pip install veritasgraph[all] # with all optional dependencies - -Quick Start: - >>> from veritasgraph import VisionRAGPipeline, VisionRAGConfig, IngestMode - >>> - >>> # "Don't Chunk. Graph." - Document-centric mode (default) - >>> config = VisionRAGConfig( - ... vision_model="llama3.2-vision:11b", - ... ingest_mode="document-centric" # whole pages/sections as nodes - ... ) - >>> pipeline = VisionRAGPipeline(config) - >>> doc = pipeline.ingest_pdf("document.pdf") - >>> - >>> # Tree-based navigation (human-like retrieval) - >>> print(pipeline.get_document_tree()) - >>> section = pipeline.navigate_to_section("Introduction") - >>> - >>> # Graph-based semantic search - >>> result = pipeline.query("What are the key findings?") - -CLI Usage: - $ veritasgraph --version - $ veritasgraph info - $ veritasgraph ingest document.pdf --ingest-mode=document-centric - $ veritasgraph init my_project - -For more information: - - Documentation: https://bibinprathap.github.io/VeritasGraph/ - - Repository: https://github.com/bibinprathap/VeritasGraph -""" - -__version__ = "0.3.0" # "Don't Chunk. Graph." - Document-centric ingestion mode -__author__ = "Bibin Prathap" -__email__ = "bibinprathap@gmail.com" -__license__ = "MIT" -__url__ = "https://github.com/bibinprathap/VeritasGraph" - - -def __getattr__(name): - """Lazy import of components to avoid import errors when dependencies not installed""" - - _vision_exports = { - "VisionRAGConfig", - "VisionModelClient", - "PDFProcessor", - "VisualElementExtractor", - "VisionKnowledgeGraph", - "VisionRAGEngine", - "VisionRAGPipeline", - } - - _model_exports = { - "VisualElement", - "DocumentPage", - "VisionDocument", - "GraphNode", - "TreeNode", - "HierarchicalStructure", - "SectionType", - "IngestMode", - } - - _tree_exports = { - "HierarchicalTreeExtractor", - "TreeQueryEngine", - } - - if name in _vision_exports: - from veritasgraph import vision - return getattr(vision, name) - - if name in _model_exports: - from veritasgraph import models - return getattr(models, name) - - if name in _tree_exports: - from veritasgraph import tree_extractor - return getattr(tree_extractor, name) - - raise AttributeError(f"module 'veritasgraph' has no attribute '{name}'") - -__all__ = [ - # Configuration - "VisionRAGConfig", - # Core clients - "VisionModelClient", - "PDFProcessor", - "VisualElementExtractor", - "VisionKnowledgeGraph", - "VisionRAGEngine", - "VisionRAGPipeline", - # Data models - "VisualElement", - "DocumentPage", - "VisionDocument", - "GraphNode", - # Hierarchical Tree Support (PageIndex-style) - "TreeNode", - "HierarchicalStructure", - "SectionType", - "HierarchicalTreeExtractor", - "TreeQueryEngine", - # "Don't Chunk. Graph." - Ingestion Modes - "IngestMode", -] diff --git a/veritasgraph/cli.py b/veritasgraph/cli.py deleted file mode 100644 index f41bc94..0000000 --- a/veritasgraph/cli.py +++ /dev/null @@ -1,473 +0,0 @@ -""" -VeritasGraph Command Line Interface -=================================== - -Entry point for the `veritasgraph` and `vg` commands. -""" - -import argparse -import sys -from pathlib import Path -from typing import Optional - -from veritasgraph import __version__ - - -def create_parser() -> argparse.ArgumentParser: - - parser = argparse.ArgumentParser( - prog="veritasgraph", - description="VeritasGraph - Enterprise-Grade Graph RAG Framework", - formatter_class=argparse.RawDescriptionHelpFormatter, - epilog=""" -Examples: - veritasgraph --version Show version - veritasgraph info Show system information - veritasgraph ingest doc.pdf Ingest a PDF document - veritasgraph query "question" Query the knowledge graph - veritasgraph serve Start the API server - -For more information, visit: https://github.com/bibinprathap/VeritasGraph - """, - ) - - parser.add_argument( - "-v", "--version", - action="version", - version=f"%(prog)s {__version__}", - ) - - parser.add_argument( - "--verbose", - action="store_true", - help="Enable verbose output", - ) - - subparsers = parser.add_subparsers(dest="command", help="Available commands") - - # Demo command - demo_parser = subparsers.add_parser("demo", help="Run VeritasGraph demo in different modes") - demo_parser.add_argument( - "--mode", - type=str, - choices=["lite", "local"], - default="lite", - help="Demo mode: 'lite' (cloud APIs, zero setup) or 'local' (offline with Ollama)" - ) - demo_parser.add_argument( - "--model", - type=str, - default=None, - help="Model to use in local mode (e.g., llama3.2)" - ) - - # Start command - start_parser = subparsers.add_parser("start", help="Start the complete GraphRAG pipeline") - start_parser.add_argument( - "--mode", - type=str, - choices=["full"], - default="full", - help="Start mode: 'full' (complete GraphRAG pipeline)" - ) - - # Info command - info_parser = subparsers.add_parser("info", help="Show system information") - info_parser.add_argument( - "--check-deps", - action="store_true", - help="Check if all dependencies are installed", - ) - - # Ingest command - ingest_parser = subparsers.add_parser("ingest", help="Ingest documents into knowledge graph") - ingest_parser.add_argument( - "source", - type=str, - help="Path to PDF file, directory, or URL to ingest", - ) - ingest_parser.add_argument( - "--output-dir", - type=str, - default="./veritasgraph_output", - help="Output directory for processed data", - ) - ingest_parser.add_argument( - "--vision-model", - type=str, - default="llama3.2-vision:11b", - help="Vision model to use for document analysis", - ) - ingest_parser.add_argument( - "--ingest-mode", - type=str, - choices=["chunk", "document-centric", "page", "section", "auto"], - default="document-centric", - help="Ingestion strategy: 'chunk' (traditional 500-token chunks), " - "'document-centric' (whole pages/sections as nodes - recommended), " - "'page' (each page is one node), 'section' (each section is one node), " - "'auto' (automatically choose based on document). Default: document-centric", - ) - - # Query command - query_parser = subparsers.add_parser("query", help="Query the knowledge graph") - query_parser.add_argument( - "question", - type=str, - help="Question to ask", - ) - query_parser.add_argument( - "--data-dir", - type=str, - default="./veritasgraph_output", - help="Directory containing the knowledge graph data", - ) - query_parser.add_argument( - "--method", - type=str, - choices=["local", "global", "hybrid"], - default="hybrid", - help="Query method to use", - ) - - # Serve command - serve_parser = subparsers.add_parser("serve", help="Start the API server") - serve_parser.add_argument( - "--host", - type=str, - default="127.0.0.1", - help="Host to bind to", - ) - serve_parser.add_argument( - "--port", - type=int, - default=8000, - help="Port to bind to", - ) - serve_parser.add_argument( - "--reload", - action="store_true", - help="Enable auto-reload for development", - ) - - # Init command - init_parser = subparsers.add_parser("init", help="Initialize a new VeritasGraph project") - init_parser.add_argument( - "path", - type=str, - nargs="?", - default=".", - help="Path to initialize project in", - ) - - return parser - - -def cmd_info(args: argparse.Namespace) -> int: - """Show system information.""" - print(f"VeritasGraph v{__version__}") - print("=" * 40) - print() - - print("Python Environment:") - print(f" Python: {sys.version}") - print(f" Platform: {sys.platform}") - print() - - print("Core Dependencies:") - deps = [ - ("ollama", "ollama"), - ("PIL", "pillow"), - ("pdf2image", "pdf2image"), - ("networkx", "networkx"), - ("numpy", "numpy"), - ("matplotlib", "matplotlib"), - ] - - for import_name, package_name in deps: - try: - module = __import__(import_name) - version = getattr(module, "__version__", "installed") - print(f" ✅ {package_name}: {version}") - except ImportError: - print(f" ❌ {package_name}: not installed") - - print() - print("Optional Dependencies:") - optional_deps = [ - ("graphrag", "graphrag"), - ("gradio", "gradio"), - ("pyvis", "pyvis"), - ("pandas", "pandas"), - ("tiktoken", "tiktoken"), - ] - - for import_name, package_name in optional_deps: - try: - module = __import__(import_name) - version = getattr(module, "__version__", "installed") - print(f" ✅ {package_name}: {version}") - except ImportError: - print(f" ⚪ {package_name}: not installed (optional)") - - if args.check_deps: - print() - print("Checking Ollama connection...") - try: - import ollama - client = ollama.Client() - models = client.list() - print(" ✅ Ollama is running") - if hasattr(models, 'models'): - model_names = [m.model for m in models.models] - else: - model_names = [m.get('name', '') for m in models.get('models', [])] - print(f" Available models: {', '.join(model_names[:5])}") - if len(model_names) > 5: - print(f" ... and {len(model_names) - 5} more") - except Exception as e: - print(f" ❌ Ollama not running or not accessible: {e}") - - return 0 - - -def cmd_ingest(args: argparse.Namespace) -> int: - """Ingest documents into knowledge graph.""" - from veritasgraph import VisionRAGConfig, VisionRAGPipeline - - source = Path(args.source) - if not source.exists(): - print(f"❌ Source not found: {source}") - return 1 - - print(f"📄 Ingesting: {source}") - - # "Don't Chunk. Graph." - Document-centric mode treats whole pages/sections as nodes - config = VisionRAGConfig( - vision_model=args.vision_model, - output_dir=args.output_dir, - ingest_mode=args.ingest_mode, - ) - - pipeline = VisionRAGPipeline(config) - - if source.is_file() and source.suffix.lower() == ".pdf": - doc = pipeline.ingest_pdf(str(source)) - if doc: - print(f"✅ Successfully ingested: {doc.title}") - print(f" Pages: {len(doc.pages)}") - print(f" Output: {args.output_dir}") - else: - print("❌ Failed to ingest document") - return 1 - elif source.is_dir(): - pdfs = list(source.glob("**/*.pdf")) - print(f"Found {len(pdfs)} PDF files") - for pdf in pdfs: - doc = pipeline.ingest_pdf(str(pdf)) - if doc: - print(f" ✅ {pdf.name}") - else: - print(f" ❌ {pdf.name}") - else: - print(f"❌ Unsupported source type: {source}") - return 1 - - return 0 - - -def cmd_query(args: argparse.Namespace) -> int: - """Query the knowledge graph.""" - print(f"🔍 Query: {args.question}") - print(f" Method: {args.method}") - print(f" Data: {args.data_dir}") - print() - - # TODO: Implement query functionality - print("⚠️ Query command not yet implemented.") - print(" Use the Gradio interface or API for queries.") - return 0 - - -# --- DEMO COMMAND --- -def cmd_demo(args: argparse.Namespace) -> int: - import os - print(f"🚀 Running VeritasGraph demo") - print(f" Mode: {args.mode}") - if args.model: - print(f" Model: {args.model}") - - if args.mode == "lite": - api_key = os.environ.get("OPENAI_API_KEY") - if not api_key or api_key == "sk-...": - print("❌ Please set your OPENAI_API_KEY environment variable for lite mode.") - return 1 - print("🌐 Using cloud APIs (OpenAI/Anthropic). Zero local setup required.") - print(" Launching demo pipeline...") - # Here you would call the appropriate pipeline or server for lite mode - print("⚠️ Demo pipeline for lite mode not yet implemented. Use the Gradio UI or API.") - return 0 - elif args.mode == "local": - print("🖥️ Using local Ollama models. 100% offline.") - if args.model: - print(f" Launching with model: {args.model}") - else: - print(" Using default local model.") - # Here you would call the appropriate pipeline or server for local mode - print("⚠️ Demo pipeline for local mode not yet implemented. Use the Gradio UI or API.") - return 0 - else: - print(f"❌ Unknown demo mode: {args.mode}") - return 1 - - -# --- START COMMAND --- -def cmd_start(args: argparse.Namespace) -> int: - print(f"🚀 Starting VeritasGraph in full mode (complete GraphRAG pipeline)") - # Here you would launch the full pipeline (API server, background workers, etc.) - print("⚠️ Full pipeline startup not yet implemented. Use 'veritasgraph serve' or the Gradio UI for now.") - return 0 - - -def cmd_serve(args: argparse.Namespace) -> int: - """Start the API server.""" - print(f"🚀 Starting VeritasGraph API server...") - print(f" Host: {args.host}") - print(f" Port: {args.port}") - print() - - # TODO: Implement server startup - print("⚠️ Server command not yet fully implemented.") - print(" Use the graphrag-ollama-config/app.py for now:") - print(" cd graphrag-ollama-config && python app.py") - return 0 - - -def cmd_init(args: argparse.Namespace) -> int: - """Initialize a new VeritasGraph project.""" - project_path = Path(args.path).resolve() - - print(f"📁 Initializing VeritasGraph project in: {project_path}") - - # Create directory structure - dirs = [ - "input", - "output", - "cache", - "prompts", - ] - - for d in dirs: - (project_path / d).mkdir(parents=True, exist_ok=True) - print(f" Created: {d}/") - - # Create settings file - settings_content = """# VeritasGraph Settings -# See: https://github.com/bibinprathap/VeritasGraph - -encoding_model: cl100k_base -skip_workflows: [] -llm: - api_key: ${GRAPHRAG_API_KEY} - type: openai_chat - model: qwen3:8b - model_supports_json: true - api_base: http://localhost:11434/v1 - -embeddings: - async_mode: threaded - llm: - api_key: ${GRAPHRAG_API_KEY} - type: openai_embedding - model: nomic-embed-text:latest - api_base: http://localhost:11434/v1 - -input: - type: file - file_type: text - base_dir: "input" - -cache: - type: file - base_dir: "cache" - -storage: - type: file - base_dir: "output" - -reporting: - type: file - base_dir: "output/reports" -""" - - settings_path = project_path / "settings.yaml" - if not settings_path.exists(): - settings_path.write_text(settings_content) - print(f" Created: settings.yaml") - else: - print(f" Skipped: settings.yaml (already exists)") - - # Create .env file - env_content = """# VeritasGraph Environment Variables -GRAPHRAG_API_KEY=ollama -OLLAMA_HOST=http://localhost:11434 -""" - - env_path = project_path / ".env" - if not env_path.exists(): - env_path.write_text(env_content) - print(f" Created: .env") - else: - print(f" Skipped: .env (already exists)") - - print() - print("✅ Project initialized!") - print() - print("Next steps:") - print(" 1. Add your documents to the 'input/' directory") - print(" 2. Run: veritasgraph ingest input/") - print(" 3. Run: veritasgraph query 'Your question here'") - - return 0 - - -def main(argv: Optional[list] = None) -> int: - """Main entry point for the CLI.""" - parser = create_parser() - args = parser.parse_args(argv) - - if args.command is None: - parser.print_help() - return 0 - - commands = { - "info": cmd_info, - "ingest": cmd_ingest, - "query": cmd_query, - "serve": cmd_serve, - "init": cmd_init, - "demo": cmd_demo, - "start": cmd_start, - } - - handler = commands.get(args.command) - if handler: - try: - return handler(args) - except KeyboardInterrupt: - print("\n\nOperation cancelled.") - return 130 - except Exception as e: - if args.verbose: - import traceback - traceback.print_exc() - else: - print(f"❌ Error: {e}") - return 1 - - parser.print_help() - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/veritasgraph/models.py b/veritasgraph/models.py deleted file mode 100644 index 906bc83..0000000 --- a/veritasgraph/models.py +++ /dev/null @@ -1,275 +0,0 @@ -""" -VeritasGraph Data Models -======================== - -Core data structures for Vision-Native RAG with Hierarchical Tree Support. -Combines the power of PageIndex's tree structure with graph flexibility. -""" - -from dataclasses import dataclass, field -from typing import List, Dict, Any, Optional, Tuple -from enum import Enum - - -class IngestMode(Enum): - """ - Ingestion modes for document processing. - - 'chunk': Traditional chunking (500-token arbitrary chunks) - 'document_centric': No chunking - treats whole pages/sections as nodes - - The document_centric mode implements the "Don't Chunk. Graph." philosophy, - neutralizing the main differentiator of PageIndex-style systems. - """ - CHUNK = "chunk" # Traditional chunking approach - DOCUMENT_CENTRIC = "document-centric" # No chunking - whole pages/sections as nodes - PAGE = "page" # Each page is a single node - SECTION = "section" # Each section (from hierarchy) is a node - AUTO = "auto" # Automatically choose based on document - - -class SectionType(Enum): - """Types of document sections for hierarchical classification""" - ROOT = "root" - CHAPTER = "chapter" - SECTION = "section" - SUBSECTION = "subsection" - SUBSUBSECTION = "subsubsection" - PARAGRAPH = "paragraph" - APPENDIX = "appendix" - REFERENCE = "reference" - TOC = "table_of_contents" - UNKNOWN = "unknown" - - -@dataclass -class TreeNode: - """ - Represents a node in the hierarchical document tree. - Inspired by PageIndex's Table-of-Contents structure. - - The structure field uses hierarchical numbering (e.g., "1", "1.1", "1.1.2") - to represent the position in the document hierarchy. - """ - id: str - title: str - structure: str # Hierarchical index like "1.2.3" - level: int # Depth in tree (0=root, 1=chapter, 2=section, etc.) - section_type: SectionType = SectionType.UNKNOWN - parent_id: Optional[str] = None - children_ids: List[str] = field(default_factory=list) - start_page: Optional[int] = None - end_page: Optional[int] = None - summary: str = "" - content_preview: str = "" - metadata: Dict[str, Any] = field(default_factory=dict) - - @property - def is_leaf(self) -> bool: - """Check if this node has no children""" - return len(self.children_ids) == 0 - - @property - def depth(self) -> int: - """Return the depth based on structure numbering""" - if not self.structure or self.structure == "root": - return 0 - return len(self.structure.split('.')) - - -@dataclass -class HierarchicalStructure: - """ - Complete hierarchical tree structure of a document. - This enables PageIndex-style "human-like retrieval" through - Table-of-Contents navigation while maintaining graph flexibility. - """ - document_id: str - root_id: str - nodes: Dict[str, TreeNode] = field(default_factory=dict) - toc_detected: bool = False - toc_pages: List[int] = field(default_factory=list) - page_offset: int = 0 # Offset between logical and physical page numbers - metadata: Dict[str, Any] = field(default_factory=dict) - - def get_node(self, node_id: str) -> Optional[TreeNode]: - """Get a node by ID""" - return self.nodes.get(node_id) - - def get_root(self) -> Optional[TreeNode]: - """Get the root node""" - return self.nodes.get(self.root_id) - - def get_children(self, node_id: str) -> List[TreeNode]: - """Get all children of a node""" - node = self.nodes.get(node_id) - if not node: - return [] - return [self.nodes[cid] for cid in node.children_ids if cid in self.nodes] - - def get_parent(self, node_id: str) -> Optional[TreeNode]: - """Get the parent of a node""" - node = self.nodes.get(node_id) - if not node or not node.parent_id: - return None - return self.nodes.get(node.parent_id) - - def get_ancestors(self, node_id: str) -> List[TreeNode]: - """Get all ancestors from node to root""" - ancestors = [] - current = self.nodes.get(node_id) - while current and current.parent_id: - parent = self.nodes.get(current.parent_id) - if parent: - ancestors.append(parent) - current = parent - else: - break - return ancestors - - def get_descendants(self, node_id: str) -> List[TreeNode]: - """Get all descendants of a node (breadth-first)""" - descendants = [] - queue = list(self.get_children(node_id)) - while queue: - node = queue.pop(0) - descendants.append(node) - queue.extend(self.get_children(node.id)) - return descendants - - def get_siblings(self, node_id: str) -> List[TreeNode]: - """Get all siblings of a node (excluding itself)""" - node = self.nodes.get(node_id) - if not node or not node.parent_id: - return [] - parent = self.nodes.get(node.parent_id) - if not parent: - return [] - return [self.nodes[cid] for cid in parent.children_ids - if cid in self.nodes and cid != node_id] - - def get_path_to_root(self, node_id: str) -> List[str]: - """Get the path from a node to root as list of titles""" - path = [] - current = self.nodes.get(node_id) - while current: - path.append(current.title) - if current.parent_id: - current = self.nodes.get(current.parent_id) - else: - break - return list(reversed(path)) - - def get_nodes_at_level(self, level: int) -> List[TreeNode]: - """Get all nodes at a specific depth level""" - return [node for node in self.nodes.values() if node.level == level] - - def get_section_for_page(self, page_number: int) -> Optional[TreeNode]: - """Find the most specific section containing a page""" - candidates = [] - for node in self.nodes.values(): - if node.start_page and node.end_page: - if node.start_page <= page_number <= node.end_page: - candidates.append(node) - elif node.start_page and node.start_page <= page_number: - candidates.append(node) - - # Return the most specific (deepest) section - if candidates: - return max(candidates, key=lambda n: n.level) - return None - - def to_dict(self) -> Dict[str, Any]: - """Export tree structure to dictionary""" - return { - "document_id": self.document_id, - "root_id": self.root_id, - "toc_detected": self.toc_detected, - "toc_pages": self.toc_pages, - "page_offset": self.page_offset, - "nodes": { - nid: { - "id": node.id, - "title": node.title, - "structure": node.structure, - "level": node.level, - "section_type": node.section_type.value, - "parent_id": node.parent_id, - "children_ids": node.children_ids, - "start_page": node.start_page, - "end_page": node.end_page, - "summary": node.summary - } - for nid, node in self.nodes.items() - }, - "metadata": self.metadata - } - - -@dataclass -class VisualElement: - """Represents a visual element extracted from a document page""" - id: str - element_type: str # 'table', 'chart', 'diagram', 'text_region', 'image' - page_number: int - description: str - structured_data: Optional[Dict[str, Any]] = None - raw_text: Optional[str] = None - confidence: float = 0.0 - bounding_box: Optional[Tuple[int, int, int, int]] = None - image_base64: Optional[str] = None - section_id: Optional[str] = None # Link to hierarchical tree node - metadata: Dict[str, Any] = field(default_factory=dict) - - -@dataclass -class DocumentPage: - """Represents a single page from a document""" - page_number: int - image_path: str - image_base64: str - width: int - height: int - elements: List[VisualElement] = field(default_factory=list) - page_summary: str = "" - page_type: str = "unknown" - section_id: Optional[str] = None # Link to hierarchical tree node - - -@dataclass -class VisionDocument: - """Complete document with all visual analysis and hierarchical structure""" - id: str - source_path: str - title: str - pages: List[DocumentPage] = field(default_factory=list) - document_summary: str = "" - document_type: str = "unknown" - extracted_entities: List[Dict[str, Any]] = field(default_factory=list) - hierarchy: Optional[HierarchicalStructure] = None # Tree structure - metadata: Dict[str, Any] = field(default_factory=dict) - - -@dataclass -class GraphNode: - """Knowledge graph node derived from visual content""" - id: str - node_type: str - name: str - description: str - source_element_id: str - source_page: int - properties: Dict[str, Any] = field(default_factory=dict) - embedding: Optional[List[float]] = None - section_id: Optional[str] = None # Link to hierarchical tree node - - -__all__ = [ - "SectionType", - "TreeNode", - "HierarchicalStructure", - "VisualElement", - "DocumentPage", - "VisionDocument", - "GraphNode", -] diff --git a/veritasgraph/py.typed b/veritasgraph/py.typed deleted file mode 100644 index 9f5c115..0000000 --- a/veritasgraph/py.typed +++ /dev/null @@ -1 +0,0 @@ -# PEP 561 marker file - indicates this package supports type checking diff --git a/veritasgraph/tree_extractor.py b/veritasgraph/tree_extractor.py deleted file mode 100644 index cb6aef7..0000000 --- a/veritasgraph/tree_extractor.py +++ /dev/null @@ -1,689 +0,0 @@ -""" -VeritasGraph Tree Extractor -=========================== - -Hierarchical tree structure extraction for documents. -Inspired by PageIndex's Table-of-Contents (TOC) approach for "human-like retrieval". - -This module combines: -- PageIndex's TOC detection and hierarchical numbering -- Vision-based structure analysis -- Intelligent parent-child relationship building - -The result is a tree structure inside the knowledge graph that enables -both tree-based navigation AND graph-based semantic search. -""" - -import re -import json -import hashlib -from typing import List, Dict, Any, Optional, Tuple -from PIL import Image - -from veritasgraph.models import ( - TreeNode, - HierarchicalStructure, - SectionType, - DocumentPage, -) - - -class HierarchicalTreeExtractor: - """ - Extracts hierarchical tree structure from documents using vision models. - - This combines the power of: - - PageIndex's TOC-based human-like retrieval - - Graph-based flexibility for complex queries - - Key features: - - Automatic TOC detection - - Hierarchical numbering extraction (1, 1.1, 1.1.1, etc.) - - Parent-child relationship building - - Page range inference for sections - """ - - PROMPTS = { - "toc_detection": """ -Analyze this document page and determine if it contains a Table of Contents (TOC). - -Look for: -- "Table of Contents", "Contents", "Index" headings -- Numbered or bulleted lists of section titles -- Section titles with page numbers (dots or spaces leading to numbers) -- Hierarchical structure indicators (1, 1.1, 1.2, 2, 2.1, etc.) - -Return JSON: -{ - "is_toc_page": true/false, - "confidence": 0.0-1.0, - "toc_type": "explicit" | "implicit" | "none", - "has_page_numbers": true/false, - "structure_type": "numbered" | "bulleted" | "mixed" | "none", - "reasoning": "brief explanation" -} - -Note: Abstract, Summary, Notation List, Figure List, Table List are NOT Table of Contents. -""", - - "toc_extraction": """ -Extract the complete Table of Contents from this page. - -For each section entry, identify: -1. The hierarchical structure number (1, 1.1, 1.1.2, etc.) if present -2. The section title -3. The page number if visible - -Return JSON: -{ - "toc_entries": [ - { - "structure": "1" or "1.1" or null (hierarchical number as string), - "title": "Section title exactly as shown", - "page": page_number_if_visible_else_null, - "level": inferred_depth_0_to_5 - } - ], - "total_entries": count, - "has_page_numbers": true/false -} - -IMPORTANT: -- Extract titles EXACTLY as shown (preserve formatting) -- Use null for missing values, not empty strings -- Level 0 = document title, Level 1 = chapter, Level 2 = section, etc. -""", - - "page_structure_analysis": """ -Analyze the structure of this document page to identify sections and subsections. - -Look for: -- Section headers (usually larger, bold, or numbered text) -- Subsection headers (smaller than main sections but distinct from body) -- Chapter markers or numbered headings -- Any hierarchical organization of content - -Return JSON: -{ - "sections_found": [ - { - "title": "Section title as shown", - "structure_number": "1.2.3" or null, - "level": 1-5 (1=main section, 5=deepest subsection), - "section_type": "chapter" | "section" | "subsection" | "heading" | "other", - "appears_to_start_here": true/false, - "confidence": 0.0-1.0 - } - ], - "page_type": "content" | "toc" | "cover" | "appendix" | "references", - "has_clear_structure": true/false -} -""", - - "structure_continuation": """ -You are continuing to extract the hierarchical tree structure of a document. - -Previous structure extracted: -{previous_structure} - -Current page content to analyze: -[Image provided] - -Continue the structure by identifying any NEW sections that START on this page. -Maintain consistent numbering with the previous structure. - -Return JSON: -{ - "new_sections": [ - { - "structure": "x.x.x" (continuing from previous), - "title": "Section title", - "starts_on_page": page_number, - "level": depth_level, - "section_type": "chapter" | "section" | "subsection" - } - ], - "continues_previous_section": true/false, - "previous_section_title": "title if continues" or null -} -""" - } - - def __init__(self, vision_client): - """ - Initialize the tree extractor. - - Args: - vision_client: VisionModelClient instance for image analysis - """ - self.vision_client = vision_client - self.prompts = self.PROMPTS - - def extract_hierarchy( - self, - pages: List[DocumentPage], - doc_id: str - ) -> HierarchicalStructure: - """ - Extract complete hierarchical structure from document pages. - - This is the main entry point that: - 1. Detects if document has a TOC - 2. Extracts TOC structure if present - 3. Falls back to page-by-page analysis if no TOC - 4. Builds parent-child relationships - 5. Infers page ranges for sections - - Args: - pages: List of DocumentPage objects with images - doc_id: Document identifier - - Returns: - HierarchicalStructure with complete tree - """ - print(f"\n🌳 Extracting hierarchical structure for document {doc_id}...") - - # Initialize structure - hierarchy = HierarchicalStructure( - document_id=doc_id, - root_id=f"{doc_id}_root" - ) - - # Create root node - root_node = TreeNode( - id=hierarchy.root_id, - title="Document Root", - structure="root", - level=0, - section_type=SectionType.ROOT, - parent_id=None - ) - hierarchy.nodes[root_node.id] = root_node - - # Step 1: Detect TOC pages - toc_pages, toc_data = self._detect_toc_pages(pages) - hierarchy.toc_detected = len(toc_pages) > 0 - hierarchy.toc_pages = toc_pages - - if hierarchy.toc_detected: - print(f" 📑 TOC detected on pages: {toc_pages}") - # Step 2a: Extract structure from TOC - self._extract_from_toc(hierarchy, toc_data, pages, doc_id) - else: - print(f" 📄 No TOC detected, analyzing page structure...") - # Step 2b: Extract structure from page analysis - self._extract_from_pages(hierarchy, pages, doc_id) - - # Step 3: Build parent-child relationships - self._build_relationships(hierarchy) - - # Step 4: Infer page ranges - self._infer_page_ranges(hierarchy, len(pages)) - - # Step 5: Link pages to sections - self._link_pages_to_sections(hierarchy, pages) - - print(f" ✅ Extracted {len(hierarchy.nodes)} tree nodes") - print(f" 📊 Max depth: {max((n.level for n in hierarchy.nodes.values()), default=0)}") - - return hierarchy - - def _detect_toc_pages( - self, - pages: List[DocumentPage], - max_pages_to_check: int = 10 - ) -> Tuple[List[int], List[Dict]]: - """ - Detect which pages contain Table of Contents. - - Args: - pages: Document pages to analyze - max_pages_to_check: Maximum pages to check for TOC - - Returns: - Tuple of (list of TOC page numbers, TOC analysis results) - """ - toc_pages = [] - toc_data = [] - consecutive_non_toc = 0 - - for i, page in enumerate(pages[:max_pages_to_check]): - if consecutive_non_toc >= 3 and toc_pages: - # Stop if we've found TOC and then had 3 non-TOC pages - break - - try: - image = Image.open(page.image_path) if page.image_path else None - if not image: - continue - - result = self.vision_client.analyze_with_json( - image, - self.prompts["toc_detection"] - ) - - if result.get("is_toc_page", False) and result.get("confidence", 0) > 0.6: - toc_pages.append(page.page_number) - toc_data.append({ - "page_number": page.page_number, - "analysis": result - }) - consecutive_non_toc = 0 - else: - if toc_pages: # Only count after we've found at least one TOC page - consecutive_non_toc += 1 - - except Exception as e: - print(f" ⚠️ Error analyzing page {page.page_number}: {e}") - continue - - return toc_pages, toc_data - - def _extract_from_toc( - self, - hierarchy: HierarchicalStructure, - toc_data: List[Dict], - pages: List[DocumentPage], - doc_id: str - ): - """ - Extract hierarchical structure from detected TOC pages. - """ - all_entries = [] - - for toc_info in toc_data: - page_num = toc_info["page_number"] - page = next((p for p in pages if p.page_number == page_num), None) - if not page: - continue - - try: - image = Image.open(page.image_path) - result = self.vision_client.analyze_with_json( - image, - self.prompts["toc_extraction"] - ) - - entries = result.get("toc_entries", []) - for entry in entries: - entry["source_toc_page"] = page_num - all_entries.extend(entries) - - except Exception as e: - print(f" ⚠️ Error extracting TOC from page {page_num}: {e}") - - # Deduplicate entries (in case TOC spans multiple pages) - seen_titles = set() - unique_entries = [] - for entry in all_entries: - title = entry.get("title", "").strip() - if title and title not in seen_titles: - seen_titles.add(title) - unique_entries.append(entry) - - # Create tree nodes from entries - for i, entry in enumerate(unique_entries): - structure = entry.get("structure") or str(i + 1) - level = entry.get("level", self._infer_level_from_structure(structure)) - - node_id = f"{doc_id}_section_{self._safe_id(entry.get('title', str(i)))}" - - node = TreeNode( - id=node_id, - title=entry.get("title", f"Section {i+1}"), - structure=structure, - level=level, - section_type=self._infer_section_type(level), - start_page=entry.get("page"), - metadata={ - "from_toc": True, - "toc_page": entry.get("source_toc_page") - } - ) - - hierarchy.nodes[node_id] = node - - def _extract_from_pages( - self, - hierarchy: HierarchicalStructure, - pages: List[DocumentPage], - doc_id: str - ): - """ - Extract hierarchical structure by analyzing each page. - Used when no TOC is detected. - """ - current_structure_counters = [0] * 6 # Support up to 6 levels deep - - for page in pages: - try: - image = Image.open(page.image_path) if page.image_path else None - if not image: - continue - - result = self.vision_client.analyze_with_json( - image, - self.prompts["page_structure_analysis"] - ) - - sections = result.get("sections_found", []) - - for section in sections: - if not section.get("appears_to_start_here", False): - continue - if section.get("confidence", 0) < 0.5: - continue - - level = section.get("level", 1) - title = section.get("title", "").strip() - - if not title: - continue - - # Generate structure number if not provided - structure = section.get("structure_number") - if not structure: - # Auto-generate hierarchical number - current_structure_counters[level] += 1 - # Reset deeper levels - for l in range(level + 1, len(current_structure_counters)): - current_structure_counters[l] = 0 - structure = ".".join( - str(current_structure_counters[l]) - for l in range(1, level + 1) - if current_structure_counters[l] > 0 - ) - - node_id = f"{doc_id}_section_{self._safe_id(title)}" - - # Skip if we already have this section - if node_id in hierarchy.nodes: - continue - - node = TreeNode( - id=node_id, - title=title, - structure=structure, - level=level, - section_type=self._infer_section_type(level), - start_page=page.page_number, - metadata={ - "from_toc": False, - "detected_on_page": page.page_number, - "confidence": section.get("confidence", 0.5) - } - ) - - hierarchy.nodes[node_id] = node - - except Exception as e: - print(f" ⚠️ Error analyzing page {page.page_number}: {e}") - - def _build_relationships(self, hierarchy: HierarchicalStructure): - """ - Build parent-child relationships based on structure numbering. - - For example: - - "1" is parent of "1.1", "1.2" - - "1.1" is parent of "1.1.1", "1.1.2" - """ - root = hierarchy.nodes.get(hierarchy.root_id) - if not root: - return - - # Sort nodes by structure for proper ordering - sorted_nodes = sorted( - [n for n in hierarchy.nodes.values() if n.id != hierarchy.root_id], - key=lambda x: self._structure_sort_key(x.structure) - ) - - for node in sorted_nodes: - parent_structure = self._get_parent_structure(node.structure) - - if parent_structure == "root" or not parent_structure: - # Top-level section - parent is root - node.parent_id = hierarchy.root_id - root.children_ids.append(node.id) - else: - # Find parent by structure - parent = next( - (n for n in hierarchy.nodes.values() - if n.structure == parent_structure), - None - ) - - if parent: - node.parent_id = parent.id - if node.id not in parent.children_ids: - parent.children_ids.append(node.id) - else: - # Parent not found, attach to root - node.parent_id = hierarchy.root_id - root.children_ids.append(node.id) - - def _infer_page_ranges(self, hierarchy: HierarchicalStructure, total_pages: int): - """ - Infer end pages for each section based on the start of the next section. - """ - # Get all nodes sorted by start page - nodes_with_pages = sorted( - [n for n in hierarchy.nodes.values() - if n.start_page is not None and n.id != hierarchy.root_id], - key=lambda x: (x.start_page, x.level) - ) - - for i, node in enumerate(nodes_with_pages): - # Find the next section at same or higher level - next_start = total_pages + 1 - for j in range(i + 1, len(nodes_with_pages)): - next_node = nodes_with_pages[j] - if next_node.level <= node.level: - next_start = next_node.start_page - break - - node.end_page = next_start - 1 - - # Set root node range - root = hierarchy.nodes.get(hierarchy.root_id) - if root: - root.start_page = 1 - root.end_page = total_pages - - def _link_pages_to_sections( - self, - hierarchy: HierarchicalStructure, - pages: List[DocumentPage] - ): - """ - Link each page to its most specific containing section. - """ - for page in pages: - section = hierarchy.get_section_for_page(page.page_number) - if section: - page.section_id = section.id - - # Helper methods - - def _safe_id(self, text: str) -> str: - """Generate a safe ID from text""" - if not text: - return hashlib.md5(str(id(text)).encode()).hexdigest()[:8] - # Remove special characters and limit length - safe = re.sub(r'[^\w\s-]', '', text) - safe = re.sub(r'[-\s]+', '_', safe)[:50] - return safe or hashlib.md5(text.encode()).hexdigest()[:8] - - def _infer_level_from_structure(self, structure: str) -> int: - """Infer tree level from structure number like '1.2.3'""" - if not structure or structure == "root": - return 0 - parts = structure.split('.') - return len(parts) - - def _infer_section_type(self, level: int) -> SectionType: - """Infer section type from level""" - level_map = { - 0: SectionType.ROOT, - 1: SectionType.CHAPTER, - 2: SectionType.SECTION, - 3: SectionType.SUBSECTION, - 4: SectionType.SUBSUBSECTION, - } - return level_map.get(level, SectionType.PARAGRAPH) - - def _get_parent_structure(self, structure: str) -> str: - """Get parent structure number from child""" - if not structure or structure == "root": - return "root" - parts = structure.split('.') - if len(parts) <= 1: - return "root" - return '.'.join(parts[:-1]) - - def _structure_sort_key(self, structure: str) -> Tuple: - """Create sort key for structure numbers like '1.2.3'""" - if not structure or structure == "root": - return (0,) - try: - parts = structure.split('.') - return tuple(int(p) if p.isdigit() else ord(p[0]) for p in parts) - except: - return (999,) - - -class TreeQueryEngine: - """ - Query engine optimized for tree-based retrieval. - - Enables "human-like retrieval" by: - - Finding sections by title/topic - - Navigating up/down the tree - - Getting context from parent sections - - Finding related sibling sections - """ - - def __init__(self, hierarchy: HierarchicalStructure): - self.hierarchy = hierarchy - - def find_sections_by_title( - self, - query: str, - fuzzy: bool = True - ) -> List[TreeNode]: - """ - Find sections matching a title query. - - Args: - query: Search query for section titles - fuzzy: Whether to do fuzzy matching - - Returns: - List of matching TreeNode objects - """ - query_lower = query.lower() - matches = [] - - for node in self.hierarchy.nodes.values(): - title_lower = node.title.lower() - - if fuzzy: - # Fuzzy match - check if query words appear in title - query_words = query_lower.split() - if all(word in title_lower for word in query_words): - matches.append(node) - elif query_lower in title_lower: - matches.append(node) - else: - # Exact match - if query_lower == title_lower: - matches.append(node) - - # Sort by level (shallower first) then by structure - return sorted(matches, key=lambda n: (n.level, n.structure)) - - def get_section_with_context(self, node_id: str) -> Dict[str, Any]: - """ - Get a section with full tree context. - - Returns: - Dict with section, parent chain, children, and siblings - """ - node = self.hierarchy.get_node(node_id) - if not node: - return {} - - return { - "section": node, - "breadcrumb": self.hierarchy.get_path_to_root(node_id), - "parent": self.hierarchy.get_parent(node_id), - "children": self.hierarchy.get_children(node_id), - "siblings": self.hierarchy.get_siblings(node_id), - "ancestors": self.hierarchy.get_ancestors(node_id), - "page_range": (node.start_page, node.end_page) - } - - def get_tree_view(self, max_depth: int = None) -> str: - """ - Generate a text tree view of the hierarchy. - - Args: - max_depth: Maximum depth to display - - Returns: - String representation of the tree - """ - lines = [] - root = self.hierarchy.get_root() - if not root: - return "Empty tree" - - self._build_tree_view(root, lines, "", max_depth) - return "\n".join(lines) - - def _build_tree_view( - self, - node: TreeNode, - lines: List[str], - prefix: str, - max_depth: int = None - ): - """Recursively build tree view""" - if max_depth is not None and node.level > max_depth: - return - - # Format node line - page_info = "" - if node.start_page: - if node.end_page and node.end_page != node.start_page: - page_info = f" (pp. {node.start_page}-{node.end_page})" - else: - page_info = f" (p. {node.start_page})" - - structure = f"[{node.structure}] " if node.structure and node.structure != "root" else "" - lines.append(f"{prefix}{structure}{node.title}{page_info}") - - # Add children - children = self.hierarchy.get_children(node.id) - for i, child in enumerate(children): - is_last = (i == len(children) - 1) - child_prefix = prefix + (" " if is_last else "│ ") - connector = "└── " if is_last else "├── " - - # Add connector before recursive call - if lines: - next_prefix = prefix + connector - else: - next_prefix = "" - - self._build_tree_view( - child, - lines, - prefix + (" " if is_last else "│ "), - max_depth - ) - - -__all__ = [ - "HierarchicalTreeExtractor", - "TreeQueryEngine", -] diff --git a/veritasgraph/vision.py b/veritasgraph/vision.py deleted file mode 100644 index c6ee912..0000000 --- a/veritasgraph/vision.py +++ /dev/null @@ -1,1628 +0,0 @@ -""" -VeritasGraph Vision Module -========================== - -Vision-Native RAG components for processing PDFs with multimodal LLMs. -Bypasses OCR for accurate table and chart extraction. - -Now with Hierarchical Tree Support: -- Combines PageIndex's TOC-based "human-like retrieval" -- With the flexibility of graph-based semantic search -""" - -import os -import json -import base64 -import hashlib -from io import BytesIO -from datetime import datetime -from typing import List, Dict, Any, Optional, Tuple -from dataclasses import dataclass, field - -import numpy as np -import networkx as nx -from PIL import Image - -from veritasgraph.models import ( - VisualElement, - DocumentPage, - VisionDocument, - GraphNode, - TreeNode, - HierarchicalStructure, - SectionType, - IngestMode, -) - - -@dataclass -class VisionRAGConfig: - """ - Configuration for Vision-Native RAG. - - Supports multiple ingestion modes: - - 'chunk': Traditional 500-token chunking - - 'document-centric': No chunking - whole pages/sections as nodes - - 'page': Each page becomes a single node - - 'section': Each section from hierarchy becomes a node - - 'auto': Automatically choose based on document structure - - The 'document-centric' mode implements: "Don't Chunk. Graph." - """ - - # Vision model settings - vision_model: str = "llama3.2-vision:11b" - text_model: str = "qwen3:8b" - embedding_model: str = "nomic-embed-text:latest" - - # Ollama settings - ollama_host: str = "http://localhost:11434" - - # PDF processing settings - pdf_dpi: int = 200 - max_image_size: Tuple[int, int] = (1568, 1568) - - # Extraction settings - extract_tables: bool = True - extract_charts: bool = True - extract_text_regions: bool = True - extract_diagrams: bool = True - - # Ingestion mode: "Don't Chunk. Graph." - ingest_mode: str = "document-centric" # chunk, document-centric, page, section, auto - - # Graph settings - min_confidence: float = 0.7 - - # Output paths - output_dir: str = "./vision_rag_output" - cache_dir: str = "./vision_rag_cache" - - def get_ingest_mode(self) -> IngestMode: - """Get the IngestMode enum from string config.""" - mode_map = { - "chunk": IngestMode.CHUNK, - "document-centric": IngestMode.DOCUMENT_CENTRIC, - "page": IngestMode.PAGE, - "section": IngestMode.SECTION, - "auto": IngestMode.AUTO, - } - return mode_map.get(self.ingest_mode.lower(), IngestMode.DOCUMENT_CENTRIC) - - -class VisionModelClient: - """ - Client for interacting with local multimodal models via Ollama. - Supports Llama 3.2 Vision, LLaVA, and other vision-capable models. - """ - - def __init__(self, config: VisionRAGConfig): - self.config = config - - try: - import ollama - self.client = ollama.Client(host=config.ollama_host) - self._ollama_available = True - except ImportError: - print("⚠️ ollama package not installed. Run: pip install ollama") - self._ollama_available = False - self.client = None - - if self._ollama_available: - self._verify_models() - - def _verify_models(self): - """Verify required models are available""" - try: - models = self.client.list() - # Handle different ollama library versions - if hasattr(models, 'models'): - # Newer ollama API returns object with .models attribute - model_list = models.models - available = [m.model if hasattr(m, 'model') else m.get('name', '') for m in model_list] - elif isinstance(models, dict): - # Older API returns dict with 'models' key - available = [m.get('name', m.get('model', '')) for m in models.get('models', [])] - else: - available = [] - - print(f"📋 Available models: {available}") - - vision_available = any(self.config.vision_model.split(':')[0] in m for m in available) - if not vision_available: - print(f"⚠️ Vision model '{self.config.vision_model}' not found.") - print(f" Run: ollama pull {self.config.vision_model}") - if any('llava' in m for m in available): - self.config.vision_model = 'llama3.2-vision:11b' - print(f" Using fallback: {self.config.vision_model}") - else: - print(f"✅ Vision model ready: {self.config.vision_model}") - - except Exception as e: - print(f"❌ Error connecting to Ollama: {e}") - print(" Make sure Ollama is running: ollama serve") - - def image_to_base64(self, image: Image.Image) -> str: - """Convert PIL Image to base64 string""" - if image.size[0] > self.config.max_image_size[0] or image.size[1] > self.config.max_image_size[1]: - image.thumbnail(self.config.max_image_size, Image.Resampling.LANCZOS) - - if image.mode != 'RGB': - image = image.convert('RGB') - - buffer = BytesIO() - image.save(buffer, format='JPEG', quality=85) - return base64.b64encode(buffer.getvalue()).decode('utf-8') - - def analyze_image(self, image: Image.Image, prompt: str) -> str: - """Analyze an image using the vision model.""" - if not self._ollama_available: - return "" - - image_b64 = self.image_to_base64(image) - - try: - response = self.client.chat( - model=self.config.vision_model, - messages=[ - { - 'role': 'user', - 'content': prompt, - 'images': [image_b64] - } - ], - options={'temperature': 0.1} - ) - return response['message']['content'] - except Exception as e: - print(f"❌ Vision analysis error: {e}") - return "" - - def analyze_with_json(self, image: Image.Image, prompt: str) -> Dict[str, Any]: - """Analyze image and parse JSON response""" - json_prompt = prompt + "\n\nRespond ONLY with valid JSON, no other text." - response = self.analyze_image(image, json_prompt) - - try: - if '```json' in response: - start = response.find('```json') + 7 - end = response.rfind('```') - response = response[start:end].strip() - elif '```' in response: - start = response.find('```') + 3 - end = response.rfind('```') - response = response[start:end].strip() - - return json.loads(response) - except json.JSONDecodeError: - response = response.replace("'", '"').replace('None', 'null').replace('True', 'true').replace('False', 'false') - try: - return json.loads(response) - except: - return {"raw_response": response, "parse_error": True} - - def get_embedding(self, text: str) -> List[float]: - """Get text embedding using the embedding model""" - if not self._ollama_available: - return [] - - try: - response = self.client.embeddings( - model=self.config.embedding_model, - prompt=text - ) - return response['embedding'] - except Exception as e: - print(f"❌ Embedding error: {e}") - return [] - - def text_completion(self, prompt: str) -> str: - """Get text completion using the text model""" - if not self._ollama_available: - return "" - - try: - response = self.client.chat( - model=self.config.text_model, - messages=[{'role': 'user', 'content': prompt}], - options={'temperature': 0.3} - ) - return response['message']['content'] - except Exception as e: - print(f"❌ Text completion error: {e}") - return "" - - -class PDFProcessor: - """Converts PDF documents to images for vision model processing.""" - - def __init__(self, config: VisionRAGConfig, vision_client: VisionModelClient): - self.config = config - self.vision_client = vision_client - - def pdf_to_images(self, pdf_path: str) -> List[Image.Image]: - """Convert PDF pages to PIL Images.""" - print(f"📄 Converting PDF: {pdf_path}") - - try: - import pdf2image - images = pdf2image.convert_from_path( - pdf_path, - dpi=self.config.pdf_dpi, - fmt='jpeg' - ) - print(f" ✅ Converted {len(images)} pages") - return images - except ImportError: - print("❌ pdf2image not installed. Run: pip install pdf2image") - return [] - except Exception as e: - print(f" ❌ PDF conversion error: {e}") - print(" Make sure poppler is installed:") - print(" - Windows: choco install poppler") - print(" - Mac: brew install poppler") - print(" - Linux: apt-get install poppler-utils") - return [] - - def save_page_images(self, images: List[Image.Image], doc_id: str) -> List[str]: - """Save page images to cache directory""" - paths = [] - cache_dir = os.path.join(self.config.cache_dir, doc_id) - os.makedirs(cache_dir, exist_ok=True) - - for i, img in enumerate(images): - path = os.path.join(cache_dir, f"page_{i+1:03d}.jpg") - img.save(path, 'JPEG', quality=90) - paths.append(path) - - return paths - - def process_pdf(self, pdf_path: str) -> Optional[VisionDocument]: - """Process a PDF document completely.""" - doc_id = hashlib.md5(pdf_path.encode()).hexdigest()[:12] - - images = self.pdf_to_images(pdf_path) - if not images: - return None - - image_paths = self.save_page_images(images, doc_id) - - pages = [] - for i, (img, path) in enumerate(zip(images, image_paths)): - page = DocumentPage( - page_number=i + 1, - image_path=path, - image_base64=self.vision_client.image_to_base64(img), - width=img.size[0], - height=img.size[1] - ) - pages.append(page) - - doc = VisionDocument( - id=doc_id, - source_path=pdf_path, - title=os.path.basename(pdf_path), - pages=pages, - metadata={ - "total_pages": len(pages), - "processed_at": datetime.now().isoformat() - } - ) - - return doc - - -class VisualElementExtractor: - """ - Extracts structured information from document pages using vision models. - This is the core of Vision-Native RAG - no OCR needed! - """ - - PROMPTS = { - "page_classification": """ -Analyze this document page and classify it. - -Return JSON with: -{ - "page_type": "cover" | "table_of_contents" | "content" | "chart_heavy" | "table_heavy" | "appendix" | "references", - "has_tables": true/false, - "has_charts": true/false, - "has_diagrams": true/false, - "has_images": true/false, - "main_topic": "brief description of main content", - "confidence": 0.0-1.0 -} -""", - - "table_extraction": """ -Extract ALL tables from this page. For each table: - -1. Identify the table title/caption -2. Extract column headers -3. Extract all data rows -4. Note any footnotes or special formatting - -Return JSON: -{ - "tables": [ - { - "table_id": "table_1", - "title": "Table title if visible", - "headers": ["Column1", "Column2", ...], - "rows": [ - ["value1", "value2", ...], - ... - ], - "footnotes": ["any footnotes"], - "table_type": "financial" | "statistical" | "comparison" | "summary" | "other" - } - ] -} - -IMPORTANT: Extract EXACT values as shown. For numbers, keep formatting (commas, decimals, currency symbols). -""", - - "chart_extraction": """ -Extract ALL charts/graphs from this page. For each chart: - -1. Identify chart type (bar, line, pie, scatter, etc.) -2. Extract title and axis labels -3. Extract data points/values as accurately as possible -4. Note legends and any annotations - -Return JSON: -{ - "charts": [ - { - "chart_id": "chart_1", - "chart_type": "bar" | "line" | "pie" | "scatter" | "area" | "combination" | "other", - "title": "Chart title", - "x_axis": {"label": "X axis label", "values": ["val1", "val2", ...]}, - "y_axis": {"label": "Y axis label", "unit": "$" | "%" | "units"}, - "data_series": [ - { - "name": "Series name", - "values": [10, 20, 30, ...] - } - ], - "insights": "Key insight or trend shown", - "time_period": "if applicable" - } - ] -} - -IMPORTANT: Estimate numerical values from the chart as accurately as possible. -""", - - "key_metrics": """ -Extract all KEY METRICS and NUMBERS from this page. - -Look for: -- Financial figures (revenue, profit, costs, margins) -- Percentages and growth rates -- Dates and time periods -- Quantities and counts -- Comparisons (YoY, QoQ, vs benchmark) - -Return JSON: -{ - "metrics": [ - { - "name": "Metric name", - "value": "$1.5B" or "15%" or "1,234", - "numeric_value": 1500000000, - "unit": "USD" | "%" | "count" | "other", - "context": "What this metric represents", - "time_period": "Q4 2024" or "FY2024" if applicable, - "comparison": {"vs": "prior period", "change": "+10%"} if applicable - } - ] -} -""", - - "entity_extraction": """ -Extract all NAMED ENTITIES from this page. - -Look for: -- Company names -- Person names (executives, analysts) -- Product names -- Geographic locations -- Dates and time periods -- Technical terms - -Return JSON: -{ - "entities": [ - { - "name": "Entity name", - "type": "company" | "person" | "product" | "location" | "date" | "concept", - "context": "How it appears in the document", - "related_entities": ["other entities it's connected to"] - } - ] -} -""", - - "page_summary": """ -Provide a comprehensive summary of this document page. - -Include: -1. Main topic or section -2. Key points and findings -3. Important numbers or metrics mentioned -4. Any conclusions or recommendations -5. How this page relates to a typical document structure - -Return JSON: -{ - "summary": "Detailed summary paragraph", - "key_points": ["point 1", "point 2", ...], - "section_type": "executive_summary" | "analysis" | "data" | "conclusion" | "methodology" | "other", - "importance": "high" | "medium" | "low" -} -""" - } - - def __init__(self, vision_client: VisionModelClient, config: VisionRAGConfig): - self.vision_client = vision_client - self.config = config - self.prompts = self.PROMPTS - - def extract_page_info(self, image: Image.Image, page_num: int) -> Dict[str, Any]: - """Extract all information from a single page""" - print(f" 📖 Analyzing page {page_num}...") - - results = {} - - # Step 1: Classify the page - classification = self.vision_client.analyze_with_json( - image, - self.prompts["page_classification"] - ) - results["classification"] = classification - - # Step 2: Extract tables if present - if classification.get("has_tables", False) and self.config.extract_tables: - print(f" 📊 Extracting tables...") - tables = self.vision_client.analyze_with_json( - image, - self.prompts["table_extraction"] - ) - results["tables"] = tables.get("tables", []) - - # Step 3: Extract charts if present - if classification.get("has_charts", False) and self.config.extract_charts: - print(f" 📈 Extracting charts...") - charts = self.vision_client.analyze_with_json( - image, - self.prompts["chart_extraction"] - ) - results["charts"] = charts.get("charts", []) - - # Step 4: Extract key metrics - print(f" 🔢 Extracting metrics...") - metrics = self.vision_client.analyze_with_json( - image, - self.prompts["key_metrics"] - ) - results["metrics"] = metrics.get("metrics", []) - - # Step 5: Extract entities - print(f" 🏷️ Extracting entities...") - entities = self.vision_client.analyze_with_json( - image, - self.prompts["entity_extraction"] - ) - results["entities"] = entities.get("entities", []) - - # Step 6: Generate page summary - print(f" 📝 Generating summary...") - summary = self.vision_client.analyze_with_json( - image, - self.prompts["page_summary"] - ) - results["summary"] = summary - - return results - - def create_visual_elements(self, page_results: Dict, page_num: int) -> List[VisualElement]: - """Convert extraction results to VisualElement objects""" - elements = [] - - # Create elements for tables - for i, table in enumerate(page_results.get("tables", [])): - element = VisualElement( - id=f"page{page_num}_table_{i+1}", - element_type="table", - page_number=page_num, - description=f"Table: {table.get('title', 'Untitled')}", - structured_data=table, - confidence=0.85, - metadata={"table_type": table.get("table_type", "unknown")} - ) - elements.append(element) - - # Create elements for charts - for i, chart in enumerate(page_results.get("charts", [])): - element = VisualElement( - id=f"page{page_num}_chart_{i+1}", - element_type="chart", - page_number=page_num, - description=f"{chart.get('chart_type', 'Chart')}: {chart.get('title', 'Untitled')}", - structured_data=chart, - confidence=0.80, - metadata={ - "chart_type": chart.get("chart_type"), - "insights": chart.get("insights", "") - } - ) - elements.append(element) - - # Create elements for metrics - for i, metric in enumerate(page_results.get("metrics", [])): - element = VisualElement( - id=f"page{page_num}_metric_{i+1}", - element_type="metric", - page_number=page_num, - description=f"{metric.get('name', 'Metric')}: {metric.get('value', 'N/A')}", - structured_data=metric, - confidence=0.90, - metadata={"unit": metric.get("unit"), "context": metric.get("context")} - ) - elements.append(element) - - return elements - - -class VisionKnowledgeGraph: - """ - Builds a knowledge graph from visually extracted information. - Connects entities, metrics, tables, and charts into a queryable structure. - - Now with Hierarchical Tree Support: - - Parent (Section) -> Child (Subsection) relationships - - Tree traversal for "human-like retrieval" - - Combined tree + graph navigation - - "Don't Chunk. Graph." - Document-Centric Mode: - - Treats whole pages/sections as single nodes (not arbitrary 500-token chunks) - - Preserves document structure and context - - Better for RAG with rich visual content - """ - - def __init__(self, vision_client: VisionModelClient, ingest_mode: IngestMode = IngestMode.DOCUMENT_CENTRIC): - self.vision_client = vision_client - self.ingest_mode = ingest_mode - self.graph = nx.DiGraph() - self.nodes: Dict[str, GraphNode] = {} - self.embeddings: Dict[str, List[float]] = {} - self.hierarchies: Dict[str, HierarchicalStructure] = {} # doc_id -> hierarchy - self.content_nodes: Dict[str, str] = {} # node_id -> full text content - - def add_document(self, doc: VisionDocument): - """ - Add a processed document to the knowledge graph. - - "Don't Chunk. Graph." - The ingest_mode determines how content is stored: - - CHUNK: Traditional 500-token chunks (legacy mode) - - DOCUMENT_CENTRIC: Whole pages/sections as nodes (recommended) - - PAGE: Each page becomes a single content node - - SECTION: Each section becomes a single content node - - AUTO: Automatically choose based on document structure - """ - print(f"\n🔗 Building knowledge graph for: {doc.title}") - print(f" 📋 Ingest mode: {self.ingest_mode.value}") - - doc_node_id = f"doc_{doc.id}" - self.graph.add_node( - doc_node_id, - type="document", - title=doc.title, - pages=len(doc.pages) - ) - - # Add hierarchical tree structure if available - if doc.hierarchy: - self._add_hierarchy_to_graph(doc.hierarchy, doc_node_id, doc.id) - self.hierarchies[doc.id] = doc.hierarchy - - # Determine effective ingest mode - effective_mode = self._determine_effective_mode(doc) - - # "Don't Chunk. Graph." - Create content nodes based on ingest mode - if effective_mode in (IngestMode.DOCUMENT_CENTRIC, IngestMode.PAGE, IngestMode.SECTION): - self._add_document_centric_content(doc, doc_node_id, effective_mode) - else: - # Traditional chunking mode (legacy) - self._add_chunked_content(doc, doc_node_id) - - self._connect_related_entities(doc) - - # Summary - content_node_count = len([n for n in self.graph.nodes() if self.graph.nodes[n].get('type') == 'content']) - print(f" ✅ Graph built: {self.graph.number_of_nodes()} nodes, {self.graph.number_of_edges()} edges") - if content_node_count > 0: - print(f" 📄 Content nodes (no chunking): {content_node_count}") - if doc.hierarchy: - print(f" 🌳 Tree structure: {len(doc.hierarchy.nodes)} sections") - - def _determine_effective_mode(self, doc: VisionDocument) -> IngestMode: - """Determine the effective ingest mode, handling AUTO mode.""" - if self.ingest_mode != IngestMode.AUTO: - return self.ingest_mode - - # AUTO mode: choose based on document characteristics - if doc.hierarchy and len(doc.hierarchy.nodes) > 0: - # Document has structure - use section-based - return IngestMode.SECTION - elif len(doc.pages) <= 10: - # Small document - use page-based - return IngestMode.PAGE - else: - # Large unstructured document - fall back to section if hierarchy exists, else page - return IngestMode.PAGE - - def _add_document_centric_content( - self, - doc: VisionDocument, - doc_node_id: str, - mode: IngestMode - ): - """ - "Don't Chunk. Graph." - Add content nodes for whole pages or sections. - - Unlike traditional RAG that chunks text into arbitrary 500-token pieces, - this preserves document structure by treating complete pages or sections - as single retrievable units. - """ - print(f" 📄 Creating document-centric content nodes...") - - if mode == IngestMode.SECTION and doc.hierarchy: - # Each section becomes a content node - self._add_section_content_nodes(doc, doc_node_id) - else: - # Each page becomes a content node (PAGE or DOCUMENT_CENTRIC fallback) - self._add_page_content_nodes(doc, doc_node_id) - - def _add_page_content_nodes(self, doc: VisionDocument, doc_node_id: str): - """Add whole pages as content nodes (no chunking).""" - for page in doc.pages: - page_node_id = f"page_{doc.id}_{page.page_number}" - - # Aggregate all content from the page - page_content = self._aggregate_page_content(page) - - self.graph.add_node( - page_node_id, - type="page", - content_type="whole_page", # Flag: this is a full page, not a chunk - page_number=page.page_number, - page_type=page.page_type, - summary=page.page_summary, - section_id=page.section_id, - content_length=len(page_content), - element_count=len(page.elements) - ) - - # Store the full content - self.content_nodes[page_node_id] = page_content - - self.graph.add_edge(doc_node_id, page_node_id, relation="contains_page") - - # Connect page to its section in the hierarchy - if page.section_id and doc.hierarchy: - section_node_id = f"section_{doc.id}_{page.section_id}" - if self.graph.has_node(section_node_id): - self.graph.add_edge(section_node_id, page_node_id, relation="contains_page") - - # Create embedding for the whole page - embed_text = f"Page {page.page_number}: {page.page_summary}\n{page_content[:2000]}" - embedding = self.vision_client.get_embedding(embed_text) - if embedding: - self.embeddings[page_node_id] = embedding - - # Also add individual elements for fine-grained search - for element in page.elements: - self._add_element_to_graph(element, page_node_id, doc.id, page.section_id) - - def _add_section_content_nodes(self, doc: VisionDocument, doc_node_id: str): - """Add whole sections as content nodes (no chunking).""" - # First, add page nodes for navigation - page_content_by_section: Dict[str, List[str]] = {} - - for page in doc.pages: - page_node_id = f"page_{doc.id}_{page.page_number}" - - self.graph.add_node( - page_node_id, - type="page", - page_number=page.page_number, - page_type=page.page_type, - summary=page.page_summary, - section_id=page.section_id - ) - - self.graph.add_edge(doc_node_id, page_node_id, relation="contains_page") - - # Aggregate content by section - section_id = page.section_id or "root" - if section_id not in page_content_by_section: - page_content_by_section[section_id] = [] - page_content_by_section[section_id].append(self._aggregate_page_content(page)) - - # Connect page to section - if page.section_id and doc.hierarchy: - section_node_id = f"section_{doc.id}_{page.section_id}" - if self.graph.has_node(section_node_id): - self.graph.add_edge(section_node_id, page_node_id, relation="contains_page") - - # Add elements - for element in page.elements: - self._add_element_to_graph(element, page_node_id, doc.id, page.section_id) - - # Create content nodes for each section - for section_id, tree_node in doc.hierarchy.nodes.items(): - section_node_id = f"section_{doc.id}_{section_id}" - content_node_id = f"content_{doc.id}_{section_id}" - - # Get accumulated content for this section - section_content = "\n\n".join(page_content_by_section.get(section_id, [])) - - # Also include content from child sections - children = doc.hierarchy.get_children(section_id) - for child in children: - child_content = page_content_by_section.get(child.id, []) - section_content += "\n\n" + "\n\n".join(child_content) - - if section_content.strip(): - self.graph.add_node( - content_node_id, - type="content", - content_type="whole_section", # Flag: this is a full section, not a chunk - section_id=section_id, - section_title=tree_node.title, - section_level=tree_node.level, - content_length=len(section_content) - ) - - self.content_nodes[content_node_id] = section_content - - # Connect to section node - self.graph.add_edge(section_node_id, content_node_id, relation="has_content") - - # Create embedding for the whole section - embed_text = f"Section: {tree_node.title}\n{section_content[:2000]}" - embedding = self.vision_client.get_embedding(embed_text) - if embedding: - self.embeddings[content_node_id] = embedding - - def _aggregate_page_content(self, page: DocumentPage) -> str: - """Aggregate all content from a page into a single text block.""" - content_parts = [] - - # Add page summary - if page.page_summary: - content_parts.append(f"Summary: {page.page_summary}") - - # Add element descriptions - for element in page.elements: - if element.description: - content_parts.append(f"[{element.element_type.upper()}]: {element.description}") - - # Add structured data as text - if element.structured_data: - if element.element_type == "table": - table_text = self._table_to_text(element.structured_data) - if table_text: - content_parts.append(table_text) - elif element.element_type == "metric": - name = element.structured_data.get("name", "") - value = element.structured_data.get("value", "") - unit = element.structured_data.get("unit", "") - content_parts.append(f"Metric: {name} = {value} {unit}".strip()) - elif element.element_type == "chart": - insights = element.structured_data.get("insights", "") - if insights: - content_parts.append(f"Chart Insights: {insights}") - - return "\n".join(content_parts) - - def _table_to_text(self, table_data: Dict) -> str: - """Convert table structured data to readable text.""" - headers = table_data.get("headers", []) - rows = table_data.get("rows", []) - - if not headers and not rows: - return "" - - text_parts = [] - if headers: - text_parts.append("Table Headers: " + " | ".join(str(h) for h in headers)) - - for i, row in enumerate(rows[:10]): # Limit to 10 rows - if isinstance(row, list): - text_parts.append(f"Row {i+1}: " + " | ".join(str(cell) for cell in row)) - elif isinstance(row, dict): - text_parts.append(f"Row {i+1}: " + " | ".join(f"{k}={v}" for k, v in row.items())) - - if len(rows) > 10: - text_parts.append(f"... and {len(rows) - 10} more rows") - - return "\n".join(text_parts) - - def _add_chunked_content(self, doc: VisionDocument, doc_node_id: str): - """Legacy chunking mode - adds pages and elements without full content nodes.""" - for page in doc.pages: - page_node_id = f"page_{doc.id}_{page.page_number}" - - self.graph.add_node( - page_node_id, - type="page", - page_number=page.page_number, - page_type=page.page_type, - summary=page.page_summary, - section_id=page.section_id - ) - - self.graph.add_edge(doc_node_id, page_node_id, relation="contains_page") - - if page.section_id and doc.hierarchy: - section_node_id = f"section_{doc.id}_{page.section_id}" - if self.graph.has_node(section_node_id): - self.graph.add_edge(section_node_id, page_node_id, relation="contains_page") - - for element in page.elements: - self._add_element_to_graph(element, page_node_id, doc.id, page.section_id) - - def get_content(self, node_id: str) -> Optional[str]: - """Retrieve the full content for a content node.""" - return self.content_nodes.get(node_id) - - def _add_hierarchy_to_graph( - self, - hierarchy: HierarchicalStructure, - doc_node_id: str, - doc_id: str - ): - """ - Add hierarchical tree structure to the graph. - Creates Parent -> Child edges for tree navigation. - """ - print(f" 🌳 Adding hierarchical tree structure...") - - for node_id, tree_node in hierarchy.nodes.items(): - section_node_id = f"section_{doc_id}_{node_id}" - - # Add section node with tree metadata - self.graph.add_node( - section_node_id, - type="section", - node_type="tree_node", - title=tree_node.title, - structure=tree_node.structure, - level=tree_node.level, - section_type=tree_node.section_type.value, - start_page=tree_node.start_page, - end_page=tree_node.end_page, - summary=tree_node.summary, - is_leaf=tree_node.is_leaf - ) - - # Connect section to document - if tree_node.structure == "root": - self.graph.add_edge( - doc_node_id, - section_node_id, - relation="has_structure" - ) - - # Create embedding for section - embed_text = f"Section: {tree_node.title}" - if tree_node.summary: - embed_text += f" | {tree_node.summary}" - embedding = self.vision_client.get_embedding(embed_text) - if embedding: - self.embeddings[section_node_id] = embedding - - # Add parent-child relationships (tree edges) - for node_id, tree_node in hierarchy.nodes.items(): - if tree_node.parent_id: - parent_section_id = f"section_{doc_id}_{tree_node.parent_id}" - child_section_id = f"section_{doc_id}_{node_id}" - - if self.graph.has_node(parent_section_id): - # Primary tree relationship - self.graph.add_edge( - parent_section_id, - child_section_id, - relation="parent_of", - edge_type="tree" - ) - # Reverse edge for upward traversal - self.graph.add_edge( - child_section_id, - parent_section_id, - relation="child_of", - edge_type="tree" - ) - - # Add sibling relationships - siblings = hierarchy.get_siblings(node_id) - for sibling in siblings: - sibling_section_id = f"section_{doc_id}_{sibling.id}" - section_id = f"section_{doc_id}_{node_id}" - if not self.graph.has_edge(section_id, sibling_section_id): - self.graph.add_edge( - section_id, - sibling_section_id, - relation="sibling_of", - edge_type="tree" - ) - - def _add_element_to_graph( - self, - element: VisualElement, - page_node_id: str, - doc_id: str, - section_id: Optional[str] = None - ): - """Add a visual element to the graph""" - element_node_id = f"{doc_id}_{element.id}" - - attrs = { - "type": element.element_type, - "description": element.description, - "confidence": element.confidence, - "page_number": element.page_number, - "section_id": section_id or element.section_id - } - - if element.element_type == "table" and element.structured_data: - attrs["table_data"] = element.structured_data - attrs["headers"] = element.structured_data.get("headers", []) - attrs["row_count"] = len(element.structured_data.get("rows", [])) - - elif element.element_type == "chart" and element.structured_data: - attrs["chart_data"] = element.structured_data - attrs["chart_type"] = element.structured_data.get("chart_type") - attrs["insights"] = element.structured_data.get("insights", "") - - elif element.element_type == "metric" and element.structured_data: - attrs["metric_name"] = element.structured_data.get("name") - attrs["metric_value"] = element.structured_data.get("value") - attrs["numeric_value"] = element.structured_data.get("numeric_value") - attrs["unit"] = element.structured_data.get("unit") - - self.graph.add_node(element_node_id, **attrs) - self.graph.add_edge(page_node_id, element_node_id, relation=f"contains_{element.element_type}") - - # Connect element to its section if available - if section_id: - section_node_id = f"section_{doc_id}_{section_id}" - if self.graph.has_node(section_node_id): - self.graph.add_edge( - section_node_id, - element_node_id, - relation="contains_element" - ) - - embed_text = f"{element.element_type}: {element.description}" - if element.structured_data: - embed_text += f" | Data: {json.dumps(element.structured_data)[:500]}" - - embedding = self.vision_client.get_embedding(embed_text) - if embedding: - self.embeddings[element_node_id] = embedding - - self.nodes[element_node_id] = GraphNode( - id=element_node_id, - node_type=element.element_type, - name=element.description, - description=json.dumps(element.structured_data) if element.structured_data else "", - source_element_id=element.id, - source_page=element.page_number, - properties=attrs, - embedding=embedding, - section_id=section_id - ) - - def _connect_related_entities(self, doc: VisionDocument): - """Find and connect related entities across pages""" - metrics_by_name = {} - - for page in doc.pages: - for element in page.elements: - if element.element_type == "metric" and element.structured_data: - name = element.structured_data.get("name", "").lower() - if name: - if name not in metrics_by_name: - metrics_by_name[name] = [] - metrics_by_name[name].append(f"{doc.id}_{element.id}") - - for name, node_ids in metrics_by_name.items(): - if len(node_ids) > 1: - for i in range(len(node_ids) - 1): - self.graph.add_edge( - node_ids[i], - node_ids[i+1], - relation="same_metric_different_page" - ) - - def semantic_search(self, query: str, top_k: int = 5) -> List[Tuple[str, float, Dict]]: - """Search graph nodes by semantic similarity""" - query_embedding = self.vision_client.get_embedding(query) - if not query_embedding: - return [] - - similarities = [] - for node_id, embedding in self.embeddings.items(): - sim = self._cosine_similarity(query_embedding, embedding) - node_data = dict(self.graph.nodes[node_id]) - similarities.append((node_id, sim, node_data)) - - similarities.sort(key=lambda x: x[1], reverse=True) - return similarities[:top_k] - - def _cosine_similarity(self, a: List[float], b: List[float]) -> float: - """Calculate cosine similarity between two vectors""" - a = np.array(a) - b = np.array(b) - return float(np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))) - - def get_context_for_query(self, query: str, max_tokens: int = 4000) -> str: - """Get relevant context for a query, including hierarchical tree context""" - results = self.semantic_search(query, top_k=10) - - context_parts = [] - seen_sections = set() - - for node_id, score, data in results: - if score < 0.5: - continue - - # Get hierarchical context if available - section_context = "" - section_id = data.get("section_id") - if section_id: - section_context = self._get_section_breadcrumb(node_id, section_id) - seen_sections.add(section_id) - - context = f"\n[{data['type'].upper()} | Page {data.get('page_number', 'N/A')} | Relevance: {score:.2f}]" - if section_context: - context += f"\n📍 Location: {section_context}" - context += f"\n{data.get('description', '')}\n" - - if 'table_data' in data: - table = data['table_data'] - context += f"Headers: {table.get('headers', [])}\n" - context += f"Rows: {len(table.get('rows', []))} data rows\n" - for row in table.get('rows', [])[:3]: - context += f" {row}\n" - - elif 'chart_data' in data: - chart = data['chart_data'] - context += f"Chart Type: {chart.get('chart_type')}\n" - context += f"Insight: {chart.get('insights', '')}\n" - - elif 'metric_value' in data: - context += f"Value: {data['metric_value']}\n" - context += f"Unit: {data.get('unit', 'N/A')}\n" - - context_parts.append(context) - - return "\n---\n".join(context_parts) - - def _get_section_breadcrumb(self, node_id: str, section_id: str) -> str: - """Get the hierarchical path to a section""" - # Extract doc_id from node_id - parts = node_id.split('_') - if len(parts) < 2: - return "" - - doc_id = parts[0] - hierarchy = self.hierarchies.get(doc_id) - if not hierarchy: - return "" - - path = hierarchy.get_path_to_root(section_id) - if path: - return " > ".join(path) - return "" - - def get_tree_context(self, doc_id: str, section_title: str) -> Dict[str, Any]: - """ - Get tree-based context for a specific section. - - This enables "human-like retrieval" by navigating the document tree. - """ - hierarchy = self.hierarchies.get(doc_id) - if not hierarchy: - return {"error": "No hierarchy found for document"} - - # Find section by title - matching_nodes = [ - n for n in hierarchy.nodes.values() - if section_title.lower() in n.title.lower() - ] - - if not matching_nodes: - return {"error": f"Section '{section_title}' not found"} - - node = matching_nodes[0] - - return { - "section": { - "title": node.title, - "structure": node.structure, - "level": node.level, - "pages": f"{node.start_page}-{node.end_page}" if node.start_page else "N/A" - }, - "breadcrumb": hierarchy.get_path_to_root(node.id), - "parent": self._node_summary(hierarchy.get_parent(node.id)), - "children": [self._node_summary(c) for c in hierarchy.get_children(node.id)], - "siblings": [self._node_summary(s) for s in hierarchy.get_siblings(node.id)] - } - - def _node_summary(self, node: Optional[TreeNode]) -> Optional[Dict]: - """Get summary dict for a tree node""" - if not node: - return None - return { - "title": node.title, - "structure": node.structure, - "pages": f"{node.start_page}-{node.end_page}" if node.start_page else "N/A" - } - - def get_section_contents(self, doc_id: str, section_id: str) -> List[Dict]: - """ - Get all content elements within a section. - - This allows retrieving everything under a specific part of the tree. - """ - section_node_id = f"section_{doc_id}_{section_id}" - if not self.graph.has_node(section_node_id): - return [] - - contents = [] - # Get direct children (elements and pages) - for _, target, edge_data in self.graph.out_edges(section_node_id, data=True): - node_data = dict(self.graph.nodes[target]) - if node_data.get("type") in ["table", "chart", "metric", "page"]: - contents.append({ - "id": target, - "type": node_data.get("type"), - "description": node_data.get("description", node_data.get("summary", "")), - "page": node_data.get("page_number") - }) - - return contents - - def visualize(self, figsize=(15, 10), show_tree: bool = True): - """Visualize the knowledge graph with hierarchical tree structure""" - import matplotlib.pyplot as plt - - plt.figure(figsize=figsize) - - color_map = { - 'document': '#FF6B6B', - 'section': '#9B59B6', # Purple for tree nodes - 'page': '#4ECDC4', - 'table': '#45B7D1', - 'chart': '#96CEB4', - 'metric': '#FFEAA7', - 'entity': '#DDA0DD' - } - - colors = [color_map.get(self.graph.nodes[n].get('type', 'entity'), '#888888') - for n in self.graph.nodes()] - - # Use hierarchical layout if tree structure exists - if show_tree and any(self.graph.nodes[n].get('type') == 'section' for n in self.graph.nodes()): - try: - # Try to use hierarchical layout - pos = nx.nx_agraph.graphviz_layout(self.graph, prog='dot') - except: - pos = nx.spring_layout(self.graph, k=2, iterations=50) - else: - pos = nx.spring_layout(self.graph, k=2, iterations=50) - - # Draw tree edges differently - tree_edges = [(u, v) for u, v, d in self.graph.edges(data=True) - if d.get('edge_type') == 'tree'] - other_edges = [(u, v) for u, v, d in self.graph.edges(data=True) - if d.get('edge_type') != 'tree'] - - # Draw other edges first (lighter) - nx.draw_networkx_edges(self.graph, pos, edgelist=other_edges, - edge_color='#CCCCCC', arrows=True, arrowsize=10, - alpha=0.5) - - # Draw tree edges (stronger) - nx.draw_networkx_edges(self.graph, pos, edgelist=tree_edges, - edge_color='#9B59B6', arrows=True, arrowsize=15, - width=2, alpha=0.8) - - # Draw nodes - nx.draw_networkx_nodes(self.graph, pos, node_color=colors, node_size=500) - - for node_type, color in color_map.items(): - plt.scatter([], [], c=color, label=node_type, s=100) - plt.legend(loc='upper left') - - plt.title("Vision-Native Knowledge Graph with Hierarchical Tree") - plt.tight_layout() - plt.show() - - def get_tree_visualization(self, doc_id: str) -> str: - """Get ASCII tree visualization for a document""" - hierarchy = self.hierarchies.get(doc_id) - if not hierarchy: - return "No hierarchy found for document" - - from veritasgraph.tree_extractor import TreeQueryEngine - engine = TreeQueryEngine(hierarchy) - return engine.get_tree_view() - - -class VisionRAGEngine: - """ - Complete Vision-Native RAG engine. - Combines visual extraction, knowledge graph, and LLM reasoning. - """ - - def __init__( - self, - vision_client: VisionModelClient, - knowledge_graph: VisionKnowledgeGraph, - config: VisionRAGConfig - ): - self.vision_client = vision_client - self.kg = knowledge_graph - self.config = config - - def query(self, question: str, include_reasoning: bool = True) -> Dict[str, Any]: - """Answer a question using Vision-Native RAG.""" - print(f"\n🔍 Processing query: {question}") - - context = self.kg.get_context_for_query(question) - - if not context: - return { - "answer": "I couldn't find relevant information in the documents.", - "confidence": 0.0, - "sources": [] - } - - search_results = self.kg.semantic_search(question, top_k=5) - - answer_prompt = f""" -You are an expert analyst answering questions about documents. -Answer based ONLY on the provided context from visual document analysis. - -Question: {question} - -Context from Document Analysis: -{context} - -Instructions: -1. Answer the question directly and specifically -2. Cite page numbers and specific data points -3. If the information involves tables or charts, describe what they show -4. If you cannot answer from the context, say so clearly -5. Be precise with numbers and metrics - -Answer: -""" - - answer = self.vision_client.text_completion(answer_prompt) - - verification = self._verify_answer(question, answer, context) - - result = { - "answer": answer, - "confidence": verification.get("confidence", 0.8), - "verified": verification.get("is_grounded", True), - "sources": [ - { - "node_id": node_id, - "type": data.get("type"), - "page": data.get("page_number"), - "relevance": f"{score:.2f}", - "description": data.get("description", "")[:100] - } - for node_id, score, data in search_results[:3] - ] - } - - if include_reasoning: - result["reasoning"] = { - "context_retrieved": len(search_results), - "verification": verification - } - - return result - - def _verify_answer(self, question: str, answer: str, context: str) -> Dict[str, Any]: - """Verify answer is grounded in context""" - verify_prompt = f""" -Verify if this answer is accurate and grounded in the context. - -Question: {question} -Answer: {answer} -Context: {context[:2000]} - -Return JSON: -{{ - "is_grounded": true/false, - "confidence": 0.0-1.0, - "issues": ["list any problems"] -}} -""" - - result = self.vision_client.text_completion(verify_prompt) - - try: - if '```json' in result: - result = result.split('```json')[1].split('```')[0] - return json.loads(result) - except: - return {"is_grounded": True, "confidence": 0.7, "issues": []} - - def query_with_image(self, question: str, image: Image.Image) -> Dict[str, Any]: - """Answer a question about a specific image/page.""" - analysis_prompt = f""" -Analyze this document image to answer the question. - -Question: {question} - -Provide a detailed answer based on what you can see in the image. -If the image contains tables or charts, extract and cite the specific data. -""" - - answer = self.vision_client.analyze_image(image, analysis_prompt) - - return { - "answer": answer, - "confidence": 0.85, - "source": "direct_image_analysis" - } - - -class VisionRAGPipeline: - """ - Complete end-to-end Vision-Native RAG Pipeline. - - Now with Hierarchical Tree Support: - - Combines PageIndex's TOC-based "human-like retrieval" - - With the flexibility of graph-based semantic search - - "The Power of PageIndex's Tree + The Flexibility of a Graph" - """ - - def __init__(self, config: VisionRAGConfig = None, extract_tree: bool = True): - self.config = config or VisionRAGConfig() - self.extract_tree = extract_tree - - os.makedirs(self.config.output_dir, exist_ok=True) - os.makedirs(self.config.cache_dir, exist_ok=True) - - self.vision_client = VisionModelClient(self.config) - self.pdf_processor = PDFProcessor(self.config, self.vision_client) - self.extractor = VisualElementExtractor(self.vision_client, self.config) - - # "Don't Chunk. Graph." - Pass ingest mode to knowledge graph - self.knowledge_graph = VisionKnowledgeGraph( - self.vision_client, - ingest_mode=self.config.get_ingest_mode() - ) - self.rag_engine = VisionRAGEngine(self.vision_client, self.knowledge_graph, self.config) - - # Tree extractor for hierarchical structure - if self.extract_tree: - from veritasgraph.tree_extractor import HierarchicalTreeExtractor - self.tree_extractor = HierarchicalTreeExtractor(self.vision_client) - else: - self.tree_extractor = None - - self.documents: List[VisionDocument] = [] - - def ingest_pdf(self, pdf_path: str, extract_tree: bool = None) -> Optional[VisionDocument]: - """ - Ingest a PDF document through the complete pipeline. - - Args: - pdf_path: Path to the PDF file - extract_tree: Override default tree extraction setting - """ - print(f"\n{'='*60}") - print(f"📥 INGESTING: {pdf_path}") - print(f"{'='*60}") - - should_extract_tree = extract_tree if extract_tree is not None else self.extract_tree - - doc = self.pdf_processor.process_pdf(pdf_path) - if not doc: - print("❌ Failed to process PDF") - return None - - # Step 1: Extract hierarchical tree structure - if should_extract_tree and self.tree_extractor: - print(f"\n🌳 Extracting hierarchical tree structure...") - doc.hierarchy = self.tree_extractor.extract_hierarchy(doc.pages, doc.id) - if doc.hierarchy: - print(f" ✅ Tree structure extracted: {len(doc.hierarchy.nodes)} sections") - if doc.hierarchy.toc_detected: - print(f" 📑 TOC found on pages: {doc.hierarchy.toc_pages}") - - # Step 2: Analyze each page for visual elements - print(f"\n🔬 Analyzing {len(doc.pages)} pages...") - - for page in doc.pages: - image = Image.open(page.image_path) - - page_results = self.extractor.extract_page_info(image, page.page_number) - - page.page_type = page_results.get("classification", {}).get("page_type", "unknown") - page.page_summary = page_results.get("summary", {}).get("summary", "") - - page.elements = self.extractor.create_visual_elements(page_results, page.page_number) - - # Link elements to their containing section - if doc.hierarchy and page.section_id: - for element in page.elements: - element.section_id = page.section_id - - for entity in page_results.get("entities", []): - doc.extracted_entities.append({ - **entity, - "source_page": page.page_number, - "section_id": page.section_id - }) - - # Step 3: Add to knowledge graph (with tree structure) - self.knowledge_graph.add_document(doc) - - self.documents.append(doc) - - # Step 4: Generate document summary - doc.document_summary = self._generate_document_summary(doc) - - # Print summary - print(f"\n✅ Document ingested successfully!") - print(f" 📄 Pages: {len(doc.pages)}") - if doc.hierarchy: - print(f" 🌳 Sections: {len(doc.hierarchy.nodes)}") - max_depth = max((n.level for n in doc.hierarchy.nodes.values()), default=0) - print(f" 📊 Tree depth: {max_depth} levels") - print(f" 📊 Tables: {sum(len([e for e in p.elements if e.element_type == 'table']) for p in doc.pages)}") - print(f" 📈 Charts: {sum(len([e for e in p.elements if e.element_type == 'chart']) for p in doc.pages)}") - print(f" 🔢 Metrics: {sum(len([e for e in p.elements if e.element_type == 'metric']) for p in doc.pages)}") - print(f" 🏷️ Entities: {len(doc.extracted_entities)}") - - return doc - - def get_document_tree(self, doc_id: str = None) -> str: - """ - Get ASCII tree view of document structure. - - This shows the hierarchical organization like a Table of Contents. - """ - if doc_id: - return self.knowledge_graph.get_tree_visualization(doc_id) - - # Get tree for first/only document - if self.documents: - return self.knowledge_graph.get_tree_visualization(self.documents[0].id) - - return "No documents ingested" - - def navigate_to_section(self, section_title: str, doc_id: str = None) -> Dict[str, Any]: - """ - Navigate to a section by title (tree-based retrieval). - - This enables "human-like retrieval" - finding content by navigating - the document structure like a human would use a Table of Contents. - """ - if not doc_id and self.documents: - doc_id = self.documents[0].id - - if not doc_id: - return {"error": "No document specified"} - - return self.knowledge_graph.get_tree_context(doc_id, section_title) - - return doc - - def _generate_document_summary(self, doc: VisionDocument) -> str: - """Generate overall document summary""" - page_summaries = [p.page_summary for p in doc.pages if p.page_summary] - - if not page_summaries: - return "No summary available" - - summary_prompt = f""" -Create a comprehensive summary of this document based on page summaries. - -Page Summaries: -{chr(10).join(page_summaries[:10])} - -Provide: -1. Document type and purpose -2. Key findings or data points -3. Main topics covered -4. Important conclusions -""" - - return self.vision_client.text_completion(summary_prompt) - - def query(self, question: str) -> Dict[str, Any]: - """Query the ingested documents using combined tree + graph search""" - return self.rag_engine.query(question) - - def visualize_graph(self, show_tree: bool = True): - """Visualize the knowledge graph with tree structure""" - self.knowledge_graph.visualize(show_tree=show_tree) - - def export_to_json(self, output_path: str): - """Export all extracted data to JSON, including tree structure""" - export_data = { - "documents": [ - { - "id": doc.id, - "title": doc.title, - "summary": doc.document_summary, - "hierarchy": doc.hierarchy.to_dict() if doc.hierarchy else None, - "pages": [ - { - "page_number": p.page_number, - "page_type": p.page_type, - "summary": p.page_summary, - "section_id": p.section_id, - "elements": [ - { - "id": e.id, - "type": e.element_type, - "description": e.description, - "section_id": e.section_id, - "data": e.structured_data - } - for e in p.elements - ] - } - for p in doc.pages - ], - "entities": doc.extracted_entities - } - for doc in self.documents - ], - "graph_stats": { - "nodes": self.knowledge_graph.graph.number_of_nodes(), - "edges": self.knowledge_graph.graph.number_of_edges(), - "tree_nodes": sum( - len(h.nodes) for h in self.knowledge_graph.hierarchies.values() - ) - } - } - - with open(output_path, 'w') as f: - json.dump(export_data, f, indent=2, default=str) - - print(f"✅ Exported to {output_path}") - - -__all__ = [ - "VisionRAGConfig", - "VisionModelClient", - "PDFProcessor", - "VisualElementExtractor", - "VisionKnowledgeGraph", - "VisionRAGEngine", - "VisionRAGPipeline", -] - -# Re-export tree extractor for convenience -try: - from veritasgraph.tree_extractor import HierarchicalTreeExtractor, TreeQueryEngine - __all__.extend(["HierarchicalTreeExtractor", "TreeQueryEngine"]) -except ImportError: - pass