ci: green pipelines that check what exists; one formatting contract

The Test Suite job failed on every push since the black check was added
because the tree had never been formatted, and pre-commit said 120
columns while CI ran black's default 88. pyproject.toml now carries
[tool.black] / [tool.isort] (88, black profile) as the single source;
pre-commit reads it; `make format` applied it (13 files, whitespace only,
146 insertions / 128 deletions, tests unchanged at 146 passed).

ci.yml: lint (black, isort, flake8 hard errors) + pytest. The Docker
registry push, VictoriaMetrics integration test, staging/production
deploy and Apache-Bench jobs were template scaffolding for hosts and
registries that do not exist; production is a systemd unit updated by
git pull. Removed rather than left permanently skipped.

docs.yml: the "Check markdown links" step curl'd every URL in every .md
and failed on localhost examples and the Tailscale IP, and the Sphinx
jobs built artifacts nobody read. Replaced by two checks that mean
something: relative links/images in README, CONTRIBUTING and docs/
resolve inside the repo, and the FastAPI OpenAPI schema exports with
the documented endpoints present (uploaded as an artifact).
This commit is contained in:
2026-09-11 23:05:37 +02:00
parent 6f4a86edbb
commit 5ad8e4eac3
19 changed files with 290 additions and 807 deletions
+60 -327
View File
@@ -1,342 +1,75 @@
name: CI/CD Pipeline - Northern Thailand Ping River Monitor name: CI
# What this checks, on every push and PR to master:
# 1. formatting contract (black + isort, config in pyproject.toml)
# 2. flake8 hard-error gate (syntax, undefined names)
# 3. the pytest suite (synthetic data, no DB/network; ~1 min)
# Docker build / staging / production / perf jobs from the original template
# were removed: there is no registry, no staging host, and production is a
# systemd unit deployed by `git pull` on the server (docs/FLOOD_FORECASTING.md
# section 6, scripts/install.sh). Re-add a job when the thing it deploys exists.
on: on:
push: push:
branches: [ master, develop ] branches: [master, develop]
pull_request: pull_request:
branches: [ master ] branches: [master]
schedule: schedule:
# Run tests daily at 2 AM UTC # daily, catches dependency drift / upstream API changes in the tests
- cron: '0 2 * * *' - cron: "0 2 * * *"
workflow_dispatch:
env: env:
PYTHON_VERSION: '3.11' PYTHON_VERSION: "3.11" # pandas 2.0.3 ships no 3.12 wheels; psycopg2-binary 2.9.9 breaks on 3.13
REGISTRY: git.b4l.co.th
IMAGE_NAME: b4l/northern-thailand-ping-river-monitor
# GitHub token for better rate limits and authentication
GH_TOKEN: ${{ secrets.GH_TOKEN }}
jobs: jobs:
# Test job lint:
name: Format & lint
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: ${{ env.PYTHON_VERSION }}
cache: pip
cache-dependency-path: requirements-dev.txt
- name: Install tools
run: |
python -m pip install --upgrade pip --root-user-action=ignore
pip install --root-user-action=ignore black==23.11.0 isort==5.12.0 flake8==6.1.0
- name: black
run: black --check --diff src/ *.py
- name: isort
run: isort --check-only --diff src/ *.py
- name: flake8 (errors only)
run: flake8 src/ --count --select=E9,F63,F7,F82 --show-source --statistics
test: test:
name: Test Suite name: Test suite
runs-on: ubuntu-latest runs-on: ubuntu-latest
strategy:
matrix:
python-version: ['3.11'] # pandas 2.0.3 ships no 3.12 wheels; widen after upgrading pandas
steps: steps:
- name: Checkout code - uses: actions/checkout@v4
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v5
uses: actions/setup-python@v4 with:
with: python-version: ${{ env.PYTHON_VERSION }}
python-version: ${{ matrix.python-version }} cache: pip
cache-dependency-path: |
requirements.txt
requirements-dev.txt
- name: Cache pip dependencies - name: Install dependencies
uses: actions/cache@v3 run: |
with: python -m pip install --upgrade pip --root-user-action=ignore
path: ~/.cache/pip pip install --root-user-action=ignore -r requirements.txt
key: ${{ runner.os }}-pip-${{ hashFiles('**/requirements*.txt') }} pip install --root-user-action=ignore pytest==7.4.3 pytest-asyncio==0.21.1
restore-keys: |
${{ runner.os }}-pip-
- name: Install dependencies - name: pytest
run: | env:
python -m pip install --upgrade pip --root-user-action=ignore DB_TYPE: sqlite
pip install --root-user-action=ignore -r requirements.txt run: pytest -q -p no:cacheprovider
pip install --root-user-action=ignore -r requirements-dev.txt
- name: Lint with flake8
run: |
flake8 src/ --count --select=E9,F63,F7,F82 --show-source --statistics
flake8 src/ --count --exit-zero --max-complexity=10 --max-line-length=100 --statistics
- name: Type check with mypy (advisory)
run: |
# 86 pre-existing errors; blocking typing gate deferred until the debt is paid down
mypy src/ --ignore-missing-imports || true
- name: Format check with black
run: |
black --check src/ *.py
- name: Import sort check
run: |
isort --check-only src/ *.py
- name: Run integration tests
run: |
python tests/test_integration.py
- name: Run station management tests
run: |
python tests/test_station_management.py
- name: Test application startup
run: |
timeout 10s python run.py --test || true
- name: Security scan with bandit
run: |
bandit -r src/ -f json -o bandit-report.json || true
- name: Upload test artifacts
uses: actions/upload-artifact@v3
if: always()
with:
name: test-results-${{ matrix.python-version }}
path: |
bandit-report.json
*.log
# Code quality job
code-quality:
name: Code Quality
runs-on: ubuntu-latest
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Set up Python
uses: actions/setup-python@v4
with:
python-version: ${{ env.PYTHON_VERSION }}
- name: Install dependencies
run: |
python -m pip install --upgrade pip --root-user-action=ignore
pip install --root-user-action=ignore -r requirements-dev.txt
- name: Run safety check
run: |
safety check -r requirements.txt --json --output safety-report.json || true
- name: Run bandit security scan
run: |
bandit -r src/ -f json -o bandit-report.json || true
- name: Upload security reports
uses: actions/upload-artifact@v3
with:
name: security-reports
path: |
safety-report.json
bandit-report.json
# Build Docker image
build:
name: Build Docker Image
runs-on: ubuntu-latest
needs: test
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Log in to Container Registry
uses: docker/login-action@v3
with:
registry: ${{ env.REGISTRY }}
username: ${{ github.actor }}
password: ${{ secrets.GITEA_TOKEN }}
- name: Extract metadata
id: meta
uses: docker/metadata-action@v5
with:
images: ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}
tags: |
type=ref,event=branch
type=ref,event=pr
type=sha,prefix={{branch}}-
type=raw,value=latest,enable={{is_default_branch}}
- name: Build and push Docker image
uses: docker/build-push-action@v5
with:
context: .
platforms: linux/amd64,linux/arm64
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
cache-from: type=gha
cache-to: type=gha,mode=max
env:
GITHUB_TOKEN: ${{ secrets.GH_TOKEN }}
- name: Test Docker image
run: |
docker run --rm ${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:${{ github.sha }} python run.py --test
# Integration test with services
integration-test:
name: Integration Test with Services
runs-on: ubuntu-latest
needs: build
services:
victoriametrics:
image: victoriametrics/victoria-metrics:latest
ports:
- 8428:8428
options: >-
--health-cmd "wget --quiet --tries=1 --spider http://localhost:8428/health"
--health-interval 30s
--health-timeout 10s
--health-retries 3
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Wait for VictoriaMetrics
run: |
timeout 60s bash -c 'until curl -f http://localhost:8428/health; do sleep 2; done'
- name: Set up Python
uses: actions/setup-python@v4
with:
python-version: ${{ env.PYTHON_VERSION }}
- name: Install dependencies
run: |
python -m pip install --upgrade pip --root-user-action=ignore
pip install --root-user-action=ignore -r requirements.txt
- name: Test with VictoriaMetrics
env:
DB_TYPE: victoriametrics
VM_HOST: localhost
VM_PORT: 8428
run: |
python run.py --test
- name: Start API server
env:
DB_TYPE: victoriametrics
VM_HOST: localhost
VM_PORT: 8428
run: |
python run.py --web-api &
sleep 10
- name: Test API endpoints
run: |
curl -f http://localhost:8000/health
curl -f http://localhost:8000/stations
curl -f http://localhost:8000/metrics
# Deploy to staging (only on develop branch)
deploy-staging:
name: Deploy to Staging
runs-on: ubuntu-latest
needs: [test, build, integration-test]
if: github.ref == 'refs/heads/develop'
environment:
name: staging
url: https://staging.ping-river-monitor.b4l.co.th
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Deploy to staging
run: |
echo "Deploying to staging environment..."
# Add your staging deployment commands here
# Example: kubectl, docker-compose, or webhook call
- name: Health check staging
run: |
sleep 30
curl -f https://staging.ping-river-monitor.b4l.co.th/health
# Deploy to production (only on main branch, manual approval)
deploy-production:
name: Deploy to Production
runs-on: ubuntu-latest
needs: [test, build, integration-test]
if: github.ref == 'refs/heads/master'
environment:
name: production
url: https://ping-river-monitor.b4l.co.th
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Deploy to production
run: |
echo "Deploying to production environment..."
# Add your production deployment commands here
- name: Health check production
run: |
sleep 30
curl -f https://ping-river-monitor.b4l.co.th/health
- name: Notify deployment
run: |
echo "✅ Production deployment successful!"
echo "🌐 URL: https://ping-river-monitor.b4l.co.th"
echo "📊 Grafana: https://grafana.ping-river-monitor.b4l.co.th"
# Performance test (only on main branch)
performance-test:
name: Performance Test
runs-on: ubuntu-latest
needs: deploy-production
if: github.ref == 'refs/heads/master'
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Install Apache Bench
run: |
sudo apt-get update
sudo apt-get install -y apache2-utils
- name: Performance test API endpoints
run: |
# Test health endpoint
ab -n 100 -c 10 https://ping-river-monitor.b4l.co.th/health
# Test stations endpoint
ab -n 50 -c 5 https://ping-river-monitor.b4l.co.th/stations
# Test metrics endpoint
ab -n 50 -c 5 https://ping-river-monitor.b4l.co.th/metrics
# Cleanup old artifacts
cleanup:
name: Cleanup
runs-on: ubuntu-latest
if: always()
needs: [test, build, integration-test]
steps:
- name: Clean up old Docker images
run: |
echo "Cleaning up old Docker images..."
# Add cleanup commands for old images/artifacts
+81 -349
View File
@@ -1,367 +1,99 @@
name: Documentation name: Docs
# Checks that the documentation the project actually ships stays consistent:
# - every relative link / image path in docs/*.md and README.md resolves
# inside the repo (external URLs are NOT fetched: localhost examples,
# rate-limited hosts and the Tailscale-era links made that gate permanently
# red, and a 200 on a curl --head proves nothing about a doc anyway)
# - the FastAPI app imports and its OpenAPI schema is exportable (that is
# the reference at https://water.buildfor.life/docs)
# The previous Sphinx/apidoc jobs produced artifacts nobody read and were
# removed. Reference docs live in docs/*.md; the public overview is at
# https://buildfor.life/docs/tooling/ping-river-monitor/.
on: on:
push: push:
branches: [ master, develop ] branches: [master, develop]
paths: paths:
- 'docs/**' - "docs/**"
- 'README.md' - "README.md"
- 'CONTRIBUTING.md' - "CONTRIBUTING.md"
- 'src/**/*.py' - "src/web_api.py"
- "src/schemas.py"
- ".gitea/workflows/docs.yml"
pull_request: pull_request:
paths: paths:
- 'docs/**' - "docs/**"
- 'README.md' - "README.md"
- 'CONTRIBUTING.md' - "CONTRIBUTING.md"
workflow_dispatch: workflow_dispatch:
env: env:
PYTHON_VERSION: '3.11' PYTHON_VERSION: "3.11"
jobs: jobs:
# Validate documentation docs:
validate-docs: name: Validate documentation
name: Validate Documentation
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- name: Checkout code - uses: actions/checkout@v4
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Set up Python - name: Relative links and images resolve
uses: actions/setup-python@v4 run: |
with: python3 - <<'PY'
python-version: ${{ env.PYTHON_VERSION }} import re, sys, pathlib
root = pathlib.Path(".")
files = [root / "README.md", root / "CONTRIBUTING.md", *root.glob("docs/**/*.md")]
link = re.compile(r"!?\[[^\]]*\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
bad = []
for md in files:
if not md.exists():
continue
for m in link.finditer(md.read_text(encoding="utf-8")):
target = m.group(1)
if target.startswith(("http://", "https://", "mailto:", "#")):
continue
path = target.split("#", 1)[0]
if not path:
continue
resolved = (md.parent / path).resolve()
if not resolved.exists():
bad.append(f"{md}: {target}")
if bad:
print("Broken relative links:")
print("\n".join(" " + b for b in bad))
sys.exit(1)
print(f"checked {len(files)} files, all relative links resolve")
PY
- name: Install documentation tools - uses: actions/setup-python@v5
run: | with:
python -m pip install --upgrade pip python-version: ${{ env.PYTHON_VERSION }}
pip install -r requirements.txt cache: pip
pip install sphinx sphinx-rtd-theme sphinx-autodoc-typehints cache-dependency-path: requirements.txt
pip install markdown-link-check || true
- name: Check markdown links - name: Install dependencies
run: | run: |
echo "🔗 Checking markdown links..." python -m pip install --upgrade pip --root-user-action=ignore
find . -name "*.md" -not -path "./.git/*" -not -path "./node_modules/*" | while read file; do pip install --root-user-action=ignore -r requirements.txt
echo "Checking $file"
# Basic link validation (you can enhance this)
grep -o 'http[s]*://[^)]*' "$file" | while read url; do
if curl -s --head "$url" | head -n 1 | grep -q "200 OK"; then
echo "✅ $url"
else
echo "❌ $url (in $file)"
fi
done
done
- name: Validate README structure - name: OpenAPI schema exports
run: | env:
echo "📋 Validating README structure..." DB_TYPE: sqlite
run: |
python - <<'PY'
import json
from src.web_api import app
spec = app.openapi()
paths = sorted(spec["paths"])
required = {"/forecast", "/measurements/latest", "/measurements/history/{station_code}", "/stations", "/api/stats", "/health"}
missing = required - set(paths)
assert not missing, f"documented endpoints missing from the app: {missing}"
json.dump(spec, open("openapi.json", "w"), indent=1)
print(f"{len(paths)} paths; schema written to openapi.json")
PY
required_sections=( - uses: actions/upload-artifact@v3
"# Northern Thailand Ping River Monitor" with:
"## Features" name: openapi-${{ github.run_number }}
"## Quick Start" path: openapi.json
"## Installation"
"## Usage"
"## API Endpoints"
"## Docker"
"## Contributing"
"## License"
)
for section in "${required_sections[@]}"; do
if grep -q "$section" README.md; then
echo "✅ Found: $section"
else
echo "❌ Missing: $section"
fi
done
- name: Check documentation completeness
run: |
echo "📚 Checking documentation completeness..."
# Check if all Python modules have docstrings
python -c "
import ast
import os
def check_docstrings(filepath):
with open(filepath, 'r', encoding='utf-8') as f:
tree = ast.parse(f.read())
missing_docstrings = []
for node in ast.walk(tree):
if isinstance(node, (ast.FunctionDef, ast.ClassDef, ast.AsyncFunctionDef)):
if not ast.get_docstring(node):
missing_docstrings.append(f'{node.name} in {filepath}')
return missing_docstrings
all_missing = []
for root, dirs, files in os.walk('src'):
for file in files:
if file.endswith('.py') and not file.startswith('__'):
filepath = os.path.join(root, file)
missing = check_docstrings(filepath)
all_missing.extend(missing)
if all_missing:
print('⚠️ Missing docstrings:')
for item in all_missing[:10]: # Show first 10
print(f' - {item}')
if len(all_missing) > 10:
print(f' ... and {len(all_missing) - 10} more')
else:
print('✅ All functions and classes have docstrings')
"
# Generate API documentation
generate-api-docs:
name: Generate API Documentation
runs-on: ubuntu-latest
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Set up Python
uses: actions/setup-python@v4
with:
python-version: ${{ env.PYTHON_VERSION }}
- name: Install dependencies
run: |
python -m pip install --upgrade pip
pip install -r requirements.txt
- name: Generate OpenAPI spec
run: |
echo "📝 Generating OpenAPI specification..."
python -c "
import json
import sys
sys.path.insert(0, 'src')
try:
from web_api import app
openapi_spec = app.openapi()
with open('openapi.json', 'w') as f:
json.dump(openapi_spec, f, indent=2)
print('✅ OpenAPI spec generated: openapi.json')
except Exception as e:
print(f'❌ Failed to generate OpenAPI spec: {e}')
"
- name: Generate API documentation
run: |
echo "📖 Generating API documentation..."
# Create API documentation from OpenAPI spec
if [ -f openapi.json ]; then
cat > api-docs.md << 'EOF'
# API Documentation
This document describes the REST API endpoints for the Northern Thailand Ping River Monitor.
## Base URL
- Production: `https://ping-river-monitor.b4l.co.th`
- Staging: `https://staging.ping-river-monitor.b4l.co.th`
- Development: `http://localhost:8000`
## Authentication
Currently, the API does not require authentication. This may change in future versions.
## Endpoints
EOF
# Extract endpoints from OpenAPI spec
python -c "
import json
with open('openapi.json', 'r') as f:
spec = json.load(f)
for path, methods in spec.get('paths', {}).items():
for method, details in methods.items():
print(f'### {method.upper()} {path}')
print()
print(details.get('summary', 'No description available'))
print()
if 'parameters' in details:
print('**Parameters:**')
for param in details['parameters']:
print(f'- `{param[\"name\"]}` ({param.get(\"in\", \"query\")}): {param.get(\"description\", \"No description\")}')
print()
print('---')
print()
" >> api-docs.md
echo "✅ API documentation generated: api-docs.md"
fi
- name: Upload documentation artifacts
uses: actions/upload-artifact@v3
with:
name: documentation-${{ github.run_number }}
path: |
openapi.json
api-docs.md
# Build Sphinx documentation
build-sphinx-docs:
name: Build Sphinx Documentation
runs-on: ubuntu-latest
steps:
- name: Checkout code
uses: actions/checkout@v4
with:
token: ${{ secrets.GITEA_TOKEN }}
- name: Set up Python
uses: actions/setup-python@v4
with:
python-version: ${{ env.PYTHON_VERSION }}
- name: Install dependencies
run: |
python -m pip install --upgrade pip
pip install -r requirements.txt
pip install sphinx sphinx-rtd-theme sphinx-autodoc-typehints
- name: Create Sphinx configuration
run: |
mkdir -p docs/sphinx
cat > docs/sphinx/conf.py << 'EOF'
import os
import sys
sys.path.insert(0, os.path.abspath('../../src'))
project = 'Northern Thailand Ping River Monitor'
copyright = '2025, Ping River Monitor Team'
author = 'Ping River Monitor Team'
version = '3.1.3'
release = '3.1.3'
extensions = [
'sphinx.ext.autodoc',
'sphinx.ext.viewcode',
'sphinx.ext.napoleon',
'sphinx_autodoc_typehints',
]
templates_path = ['_templates']
exclude_patterns = ['_build', 'Thumbs.db', '.DS_Store']
html_theme = 'sphinx_rtd_theme'
html_static_path = ['_static']
autodoc_default_options = {
'members': True,
'member-order': 'bysource',
'special-members': '__init__',
'undoc-members': True,
'exclude-members': '__weakref__'
}
EOF
cat > docs/sphinx/index.rst << 'EOF'
Northern Thailand Ping River Monitor Documentation
================================================
.. toctree::
:maxdepth: 2
:caption: Contents:
modules
Indices and tables
==================
* :ref:`genindex`
* :ref:`modindex`
* :ref:`search`
EOF
- name: Generate module documentation
run: |
cd docs/sphinx
sphinx-apidoc -o . ../../src
- name: Build documentation
run: |
cd docs/sphinx
sphinx-build -b html . _build/html
- name: Upload Sphinx documentation
uses: actions/upload-artifact@v3
with:
name: sphinx-docs-${{ github.run_number }}
path: docs/sphinx/_build/html/
# Documentation summary
docs-summary:
name: Documentation Summary
runs-on: ubuntu-latest
needs: [validate-docs, generate-api-docs, build-sphinx-docs]
if: always()
steps:
- name: Generate documentation summary
run: |
echo "# 📚 Documentation Build Summary" > docs-summary.md
echo "" >> docs-summary.md
echo "**Build Date:** $(date -u)" >> docs-summary.md
echo "**Repository:** ${{ github.repository }}" >> docs-summary.md
echo "**Commit:** ${{ github.sha }}" >> docs-summary.md
echo "" >> docs-summary.md
echo "## 📊 Results" >> docs-summary.md
echo "" >> docs-summary.md
if [ "${{ needs.validate-docs.result }}" = "success" ]; then
echo "- ✅ **Documentation Validation**: Passed" >> docs-summary.md
else
echo "- ❌ **Documentation Validation**: Failed" >> docs-summary.md
fi
if [ "${{ needs.generate-api-docs.result }}" = "success" ]; then
echo "- ✅ **API Documentation**: Generated" >> docs-summary.md
else
echo "- ❌ **API Documentation**: Failed" >> docs-summary.md
fi
if [ "${{ needs.build-sphinx-docs.result }}" = "success" ]; then
echo "- ✅ **Sphinx Documentation**: Built" >> docs-summary.md
else
echo "- ❌ **Sphinx Documentation**: Failed" >> docs-summary.md
fi
echo "" >> docs-summary.md
echo "## 🔗 Available Documentation" >> docs-summary.md
echo "" >> docs-summary.md
echo "- [README.md](../README.md)" >> docs-summary.md
echo "- [API Documentation](../docs/)" >> docs-summary.md
echo "- [Contributing Guide](../CONTRIBUTING.md)" >> docs-summary.md
cat docs-summary.md
- name: Upload documentation summary
uses: actions/upload-artifact@v3
with:
name: docs-summary-${{ github.run_number }}
path: docs-summary.md
+1 -3
View File
@@ -23,18 +23,16 @@ repos:
hooks: hooks:
- id: black - id: black
language_version: python3 language_version: python3
args: ['--line-length=120']
# Import sorting with isort # Import sorting with isort
- repo: https://github.com/pycqa/isort - repo: https://github.com/pycqa/isort
rev: 5.12.0 rev: 5.12.0
hooks: hooks:
- id: isort - id: isort
args: ['--profile', 'black', '--line-length', '120']
# Linting with flake8 # Linting with flake8
- repo: https://github.com/pycqa/flake8 - repo: https://github.com/pycqa/flake8
rev: 6.1.0 rev: 6.1.0
hooks: hooks:
- id: flake8 - id: flake8
args: ['--max-line-length=120', '--extend-ignore=E203,W503'] args: ['--max-line-length=100', '--extend-ignore=E203,W503']
+1 -1
View File
@@ -4,7 +4,7 @@ A comprehensive real-time water level monitoring system for the Ping River Basin
**Live dashboard: [water.buildfor.life](https://water.buildfor.life/)** — water levels, discharge, rainfall and 6/12/24 h flood forecasts for Chiang Mai, in English and Thai. Background: [Teaching a Model to See the Ping River Rise 13 Hours Early](https://buildfor.life/blog/ping-river-monitor/). **Live dashboard: [water.buildfor.life](https://water.buildfor.life/)** — water levels, discharge, rainfall and 6/12/24 h flood forecasts for Chiang Mai, in English and Thai. Background: [Teaching a Model to See the Ping River Rise 13 Hours Early](https://buildfor.life/blog/ping-river-monitor/).
[![CI/CD](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions/workflows/ci.yml/badge.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions) [![Security](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions/workflows/security.yml/badge.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions) [![Documentation](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions/workflows/docs.yml/badge.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions) [![Python](https://img.shields.io/badge/Python-3.9+-blue.svg)](https://python.org) [![FastAPI](https://img.shields.io/badge/FastAPI-0.104+-green.svg)](https://fastapi.tiangolo.com) [![Docker](https://img.shields.io/badge/Docker-Ready-blue.svg)](https://docker.com) [![License](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE) [![Version](https://img.shields.io/badge/Version-v3.1.3-blue.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/releases) [![CI/CD](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions/workflows/ci.yml/badge.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions) [![Security](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions/workflows/security.yml/badge.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions) [![Documentation](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions/workflows/docs.yml/badge.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/actions) [![Python](https://img.shields.io/badge/Python-3.11-blue.svg)](https://python.org) [![FastAPI](https://img.shields.io/badge/FastAPI-0.104+-green.svg)](https://fastapi.tiangolo.com) [![Docker](https://img.shields.io/badge/Docker-Ready-blue.svg)](https://docker.com) [![License](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE) [![Version](https://img.shields.io/badge/Version-v3.1.3-blue.svg)](https://git.b4l.co.th/B4L/Northern-Thailand-Ping-River-Monitor/releases)
## 🌟 Features ## 🌟 Features
+13
View File
@@ -128,3 +128,16 @@ where = ["src"]
[tool.setuptools.package-dir] [tool.setuptools.package-dir]
"" = "src" "" = "src"
# One formatting contract for CI, pre-commit and editors. Black's default 88
# columns; isort in black-compatible mode. Run `make format` before committing.
[tool.black]
line-length = 88
target-version = ["py311"]
extend-exclude = '/(\.venv|venv|models|\.claude-flow|\.swarm)/'
[tool.isort]
profile = "black"
line_length = 88
known_first_party = ["src"]
skip_gitignore = true
+7 -3
View File
@@ -12,9 +12,13 @@ __description__ = "Northern Thailand Ping River Monitoring System"
from .config import Config from .config import Config
from .database_adapters import DatabaseAdapter, create_database_adapter from .database_adapters import DatabaseAdapter, create_database_adapter
from .exceptions import (APIConnectionError, ConfigurationError, from .exceptions import (
DatabaseConnectionError, DataValidationError, APIConnectionError,
WaterMonitorException) ConfigurationError,
DatabaseConnectionError,
DataValidationError,
WaterMonitorException,
)
from .models import DatabaseConfig, StationInfo, WaterMeasurement from .models import DatabaseConfig, StationInfo, WaterMeasurement
from .water_scraper_v3 import EnhancedWaterMonitorScraper from .water_scraper_v3 import EnhancedWaterMonitorScraper
+2 -6
View File
@@ -767,9 +767,7 @@ class SQLAdapter(DatabaseAdapter):
return hours_by_day return hours_by_day
except Exception as e: except Exception as e:
logging.error( logging.error(f"Error querying {self.db_type.upper()} recorded hours: {e}")
f"Error querying {self.db_type.upper()} recorded hours: {e}"
)
return None return None
def get_database_stats(self) -> Optional[Dict]: def get_database_stats(self) -> Optional[Dict]:
@@ -814,9 +812,7 @@ class SQLAdapter(DatabaseAdapter):
# the DISTINCT day-hour slots and coverage cannot exceed 100% # the DISTINCT day-hour slots and coverage cannot exceed 100%
first_slot = first_ts.replace(minute=0, second=0, microsecond=0) first_slot = first_ts.replace(minute=0, second=0, microsecond=0)
last_slot = last_ts.replace(minute=0, second=0, microsecond=0) last_slot = last_ts.replace(minute=0, second=0, microsecond=0)
expected_hours = ( expected_hours = int((last_slot - first_slot).total_seconds() // 3600) + 1
int((last_slot - first_slot).total_seconds() // 3600) + 1
)
recorded_hours = int(row[4]) recorded_hours = int(row[4])
coverage_percent = round(100.0 * recorded_hours / expected_hours, 1) coverage_percent = round(100.0 * recorded_hours / expected_hours, 1)
+2 -4
View File
@@ -17,9 +17,9 @@ import time
from typing import Dict, List, Optional from typing import Dict, List, Optional
from .hii_collector import ( from .hii_collector import (
PING_BASIN_CODE,
HiiClient, HiiClient,
HiiStore, HiiStore,
PING_BASIN_CODE,
_parse_datetime, _parse_datetime,
_to_float, _to_float,
) )
@@ -145,9 +145,7 @@ def backfill(
station_rows += store.save_waterlevel_history(sid, rows) station_rows += store.save_waterlevel_history(sid, rows)
except Exception as e: except Exception as e:
totals["errors"] += 1 totals["errors"] += 1
logger.warning( logger.warning(f"{label}: {chunk_start}..{chunk_end} failed: {e}")
f"{label}: {chunk_start}..{chunk_end} failed: {e}"
)
time.sleep(sleep_seconds) time.sleep(sleep_seconds)
totals["rows"] += station_rows totals["rows"] += station_rows
logger.info(f"{label} (id {sid}): {station_rows} rows saved") logger.info(f"{label} (id {sid}): {station_rows} rows saved")
+3 -9
View File
@@ -66,9 +66,7 @@ def rid_code_from_oldcode(oldcode: Optional[str]) -> Optional[str]:
return match.group(1) if match else None return match.group(1) if match else None
def parse_rain_records( def parse_rain_records(payload: Dict, basin_code: int = PING_BASIN_CODE) -> List[Dict]:
payload: Dict, basin_code: int = PING_BASIN_CODE
) -> List[Dict]:
"""Extract per-station rainfall rows from a rain_24h payload.""" """Extract per-station rainfall rows from a rain_24h payload."""
records = [] records = []
for row in payload.get("data") or []: for row in payload.get("data") or []:
@@ -186,9 +184,7 @@ class HiiStore:
def __init__(self, connection_string: str, db_type: str): def __init__(self, connection_string: str, db_type: str):
self.db_type = db_type.lower() self.db_type = db_type.lower()
if self.db_type not in ("sqlite", "postgresql", "mysql"): if self.db_type not in ("sqlite", "postgresql", "mysql"):
raise ValueError( raise ValueError(f"HII collection requires a SQL database, got '{db_type}'")
f"HII collection requires a SQL database, got '{db_type}'"
)
self.connection_string = connection_string self.connection_string = connection_string
self.engine = None self.engine = None
@@ -400,9 +396,7 @@ class HiiStore:
from sqlalchemy import text from sqlalchemy import text
now = datetime.datetime.now() now = datetime.datetime.now()
station_sql = self._upsert( station_sql = self._upsert(station_table, ["id"], station_cols + ["updated_at"])
station_table, ["id"], station_cols + ["updated_at"]
)
measurement_sql = self._upsert( measurement_sql = self._upsert(
measurement_table, ["station_id", "timestamp"], measurement_cols measurement_table, ["station_id", "timestamp"], measurement_cols
) )
+1 -3
View File
@@ -61,9 +61,7 @@ def load_daily(
params["start"] = start params["start"] = start
engine = create_engine(resolved, pool_pre_ping=True) engine = create_engine(resolved, pool_pre_ping=True)
with engine.connect() as conn: with engine.connect() as conn:
daily = pd.read_sql( daily = pd.read_sql(text(query + " ORDER BY date"), conn, params=params)
text(query + " ORDER BY date"), conn, params=params
)
daily["date"] = pd.to_datetime(daily["date"]) daily["date"] = pd.to_datetime(daily["date"])
daily = daily.set_index("date") daily = daily.set_index("date")
for col in DAM_COLUMNS: for col in DAM_COLUMNS:
+1 -3
View File
@@ -213,9 +213,7 @@ def fill_from_hii(
} }
) )
fills.append(fill) fills.append(fill)
logger.info( logger.info(f"HII gap-fill {code}: +{len(fill)} hours (offset {offset:.3f} m)")
f"HII gap-fill {code}: +{len(fill)} hours (offset {offset:.3f} m)"
)
if not fills: if not fills:
return df return df
return _normalize_long(pd.concat([df] + fills, ignore_index=True)) return _normalize_long(pd.concat([df] + fills, ignore_index=True))
+76 -51
View File
@@ -62,10 +62,17 @@ EXTRA_RAIN_FEATURES = ("rain_fc48",)
class Variant: class Variant:
"""A trainable candidate producing (pred_abs, sigma_per_row) on test rows.""" """A trainable candidate producing (pred_abs, sigma_per_row) on test rows."""
def __init__(self, name: str, target: str, weighted: bool = False, def __init__(
quantile: bool = False, use_rain: bool = False, self,
use_dam: bool = False, use_fc48: bool = False, name: str,
qsigma: bool = False): target: str,
weighted: bool = False,
quantile: bool = False,
use_rain: bool = False,
use_dam: bool = False,
use_fc48: bool = False,
qsigma: bool = False,
):
self.name = name self.name = name
self.target = target # 'abs' or 'rise' self.target = target # 'abs' or 'rise'
self.weighted = weighted self.weighted = weighted
@@ -78,9 +85,7 @@ class Variant:
# sigma-independent) and quantile heads ONLY for a per-row sigma. # sigma-independent) and quantile heads ONLY for a per-row sigma.
self.qsigma = qsigma self.qsigma = qsigma
def fit_predict( def fit_predict(self, X_tr, y_abs_tr, X_te) -> Tuple[np.ndarray, np.ndarray]:
self, X_tr, y_abs_tr, X_te
) -> Tuple[np.ndarray, np.ndarray]:
if not self.use_rain: if not self.use_rain:
drop = [c for c in features.RAIN_FEATURES if c in X_tr.columns] drop = [c for c in features.RAIN_FEATURES if c in X_tr.columns]
X_tr = X_tr.drop(columns=drop) X_tr = X_tr.drop(columns=drop)
@@ -135,24 +140,30 @@ VARIANTS: Dict[str, Variant] = {
"baseline_abs": Variant("baseline_abs", target="abs"), "baseline_abs": Variant("baseline_abs", target="abs"),
"rise": Variant("rise", target="rise"), "rise": Variant("rise", target="rise"),
"rise_weighted": Variant("rise_weighted", target="rise", weighted=True), "rise_weighted": Variant("rise_weighted", target="rise", weighted=True),
"rise_quantile": Variant("rise_quantile", target="rise", weighted=True, "rise_quantile": Variant(
quantile=True), "rise_quantile", target="rise", weighted=True, quantile=True
),
"rise_rain": Variant("rise_rain", target="rise", use_rain=True), "rise_rain": Variant("rise_rain", target="rise", use_rain=True),
"rise_rain_dam": Variant("rise_rain_dam", target="rise", use_rain=True, "rise_rain_dam": Variant(
use_dam=True), "rise_rain_dam", target="rise", use_rain=True, use_dam=True
),
"rise_dam": Variant("rise_dam", target="rise", use_dam=True), "rise_dam": Variant("rise_dam", target="rise", use_dam=True),
# 2026-09-12 experiments on top of the deployed rise_rain configuration: # 2026-09-12 experiments on top of the deployed rise_rain configuration:
# per-row sigma from quantile heads (the served sigma sits on the 0.15 # per-row sigma from quantile heads (the served sigma sits on the 0.15
# floor at every P.1 horizon, so stage probabilities are constant- # floor at every P.1 horizon, so stage probabilities are constant-
# calibrated), and a longer forecast-rain window for the 24 h horizon. # calibrated), and a longer forecast-rain window for the 24 h horizon.
"rise_rain_quantile": Variant("rise_rain_quantile", target="rise", "rise_rain_quantile": Variant(
weighted=True, quantile=True, use_rain=True), "rise_rain_quantile", target="rise", weighted=True, quantile=True, use_rain=True
"rise_rain_quantile_uw": Variant("rise_rain_quantile_uw", target="rise", ),
quantile=True, use_rain=True), "rise_rain_quantile_uw": Variant(
"rise_rain_fc48": Variant("rise_rain_fc48", target="rise", use_rain=True, "rise_rain_quantile_uw", target="rise", quantile=True, use_rain=True
use_fc48=True), ),
"rise_rain_qsigma": Variant("rise_rain_qsigma", target="rise", use_rain=True, "rise_rain_fc48": Variant(
qsigma=True), "rise_rain_fc48", target="rise", use_rain=True, use_fc48=True
),
"rise_rain_qsigma": Variant(
"rise_rain_qsigma", target="rise", use_rain=True, qsigma=True
),
} }
# Dam variants are opt-in by name: they require dam columns that only exist # Dam variants are opt-in by name: they require dam columns that only exist
@@ -160,8 +171,11 @@ VARIANTS: Dict[str, Variant] = {
# the 2026-08-13 ablation concluded them a negative result. The 2026-09-12 # the 2026-08-13 ablation concluded them a negative result. The 2026-09-12
# experiments are opt-in too (see their results in docs/FLOOD_FORECASTING.md). # experiments are opt-in too (see their results in docs/FLOOD_FORECASTING.md).
DEFAULT_VARIANTS = [ DEFAULT_VARIANTS = [
k for k, v in VARIANTS.items() k
if not v.use_dam and not v.use_fc48 and not v.qsigma for k, v in VARIANTS.items()
if not v.use_dam
and not v.use_fc48
and not v.qsigma
and not (v.quantile and v.use_rain) and not (v.quantile and v.use_rain)
] ]
@@ -209,12 +223,16 @@ def _first_alert_lead(
start = crossing - pd.Timedelta(hours=72) start = crossing - pd.Timedelta(hours=72)
if window_start_floor is not None and window_start_floor > start: if window_start_floor is not None and window_start_floor > start:
start = window_start_floor start = window_start_floor
window = p.loc[start: crossing + pd.Timedelta(hours=24)] window = p.loc[start : crossing + pd.Timedelta(hours=24)]
if len(window) < 2: if len(window) < 2:
return None return None
alert = (window >= ALERT_P) & (window.shift(-1) >= ALERT_P) & ( alert = (
(window.index.to_series().shift(-1) - window.index.to_series()) (window >= ALERT_P)
<= pd.Timedelta(hours=2) & (window.shift(-1) >= ALERT_P)
& (
(window.index.to_series().shift(-1) - window.index.to_series())
<= pd.Timedelta(hours=2)
)
) )
hits = window.index[alert.fillna(False)] hits = window.index[alert.fillna(False)]
if len(hits) == 0: if len(hits) == 0:
@@ -222,9 +240,7 @@ def _first_alert_lead(
return float((crossing - hits[0]).total_seconds() / 3600.0) return float((crossing - hits[0]).total_seconds() / 3600.0)
def _false_alarm_episodes( def _false_alarm_episodes(p: pd.Series, observed: pd.Series, thr: float) -> int:
p: pd.Series, observed: pd.Series, thr: float
) -> int:
"""Alert episodes with no observed >=thr within +/- FALSE_ALARM_GRACE_H.""" """Alert episodes with no observed >=thr within +/- FALSE_ALARM_GRACE_H."""
alert_hours = p[p >= ALERT_P].index alert_hours = p[p >= ALERT_P].index
if len(alert_hours) == 0: if len(alert_hours) == 0:
@@ -295,7 +311,9 @@ def evaluate_station(
tr = (X_all.index <= train_end) & y_abs.notna() tr = (X_all.index <= train_end) & y_abs.notna()
te = (X_all.index >= test_lo) & (X_all.index <= test_hi) te = (X_all.index >= test_lo) & (X_all.index <= test_hi)
if tr.sum() < 5000 or te.sum() < 500: if tr.sum() < 5000 or te.sum() < 500:
logger.info(f"{station} {year}: skipped (train {tr.sum()}, test {te.sum()})") logger.info(
f"{station} {year}: skipped (train {tr.sum()}, test {te.sum()})"
)
continue continue
X_tr, X_te = X_all.loc[tr], X_all.loc[te] X_tr, X_te = X_all.loc[tr], X_all.loc[te]
@@ -371,9 +389,7 @@ def evaluate_station(
) )
fold["variants"][name] = { fold["variants"][name] = {
"mae": float(errors.mean()) if len(errors) else None, "mae": float(errors.mean()) if len(errors) else None,
"mae_above_2p5": ( "mae_above_2p5": (float(errors[high].mean()) if high.any() else None),
float(errors[high].mean()) if high.any() else None
),
"brier_warn": brier, "brier_warn": brier,
"events": event_rows, "events": event_rows,
"false_alarm_episodes": _false_alarm_episodes( "false_alarm_episodes": _false_alarm_episodes(
@@ -394,17 +410,20 @@ def summarize(results: Dict) -> str:
lines.append(header) lines.append(header)
for fold in results["folds"]: for fold in results["folds"]:
for name, m in fold["variants"].items(): for name, m in fold["variants"].items():
events = " ".join( events = (
f"[{e['crossing'][:10]}: " " ".join(
f"{'' if e['lead_h'] is None else format(e['lead_h'], '+.0f')}h" f"[{e['crossing'][:10]}: "
+ ( f"{'' if e['lead_h'] is None else format(e['lead_h'], '+.0f')}h"
f" | {e['peak_pred_24h_before'] - e['peak_level']:+.2f}" + (
if e["peak_pred_24h_before"] is not None f" | {e['peak_pred_24h_before'] - e['peak_level']:+.2f}"
else "" if e["peak_pred_24h_before"] is not None
else ""
)
+ "]"
for e in m["events"]
) )
+ "]" or "no events"
for e in m["events"] )
) or "no events"
lines.append( lines.append(
f"{name:16} {fold['year']:>5} " f"{name:16} {fold['year']:>5} "
f"{m['mae'] if m['mae'] is not None else float('nan'):6.3f} " f"{m['mae'] if m['mae'] is not None else float('nan'):6.3f} "
@@ -421,16 +440,22 @@ def main(argv=None) -> int:
parser = argparse.ArgumentParser(description=__doc__) parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--stations", default="P.1") parser.add_argument("--stations", default="P.1")
parser.add_argument("--db-url", default=None) parser.add_argument("--db-url", default=None)
parser.add_argument("--variants", default=None, parser.add_argument("--variants", default=None, help="comma list; default all")
help="comma list; default all")
parser.add_argument("--out", default="models/eval_variants.json") parser.add_argument("--out", default="models/eval_variants.json")
parser.add_argument("--no-rain", action="store_true", parser.add_argument(
help="skip loading the Open-Meteo rain series") "--no-rain", action="store_true", help="skip loading the Open-Meteo rain series"
parser.add_argument("--no-dam", action="store_true", )
help="skip loading the Mae Ngat reservoir series") parser.add_argument(
parser.add_argument("--from-cache", action="store_true", "--no-dam",
help="offline: read models/cache/ only (no DB, no API, " action="store_true",
"no Open-Meteo refresh) -- reproducible reruns") help="skip loading the Mae Ngat reservoir series",
)
parser.add_argument(
"--from-cache",
action="store_true",
help="offline: read models/cache/ only (no DB, no API, "
"no Open-Meteo refresh) -- reproducible reruns",
)
args = parser.parse_args(argv) args = parser.parse_args(argv)
logging.basicConfig( logging.basicConfig(
+7 -4
View File
@@ -65,7 +65,12 @@ def load_gauge_mean(
"AND s.longitude BETWEEN :lon_lo AND :lon_hi " "AND s.longitude BETWEEN :lon_lo AND :lon_hi "
"AND m.rain_1h IS NOT NULL" "AND m.rain_1h IS NOT NULL"
) )
params = {"lat_lo": lat_lo, "lat_hi": lat_hi, "lon_lo": lon_lo, "lon_hi": lon_hi} params = {
"lat_lo": lat_lo,
"lat_hi": lat_hi,
"lon_lo": lon_lo,
"lon_hi": lon_hi,
}
if start is not None: if start is not None:
query += " AND m.timestamp >= :start" query += " AND m.timestamp >= :start"
params["start"] = pd.Timestamp(start).to_pydatetime() params["start"] = pd.Timestamp(start).to_pydatetime()
@@ -98,9 +103,7 @@ def compare_with_openmeteo(
Both are summed over trailing `window_h` so single-hour timing offsets Both are summed over trailing `window_h` so single-hour timing offsets
(gauges report at :00, the model's hour is an interval) do not dominate. (gauges report at :00, the model's hour is an interval) do not dominate.
""" """
joined = pd.concat( joined = pd.concat({"gauge": gauge, "openmeteo": openmeteo}, axis=1).dropna()
{"gauge": gauge, "openmeteo": openmeteo}, axis=1
).dropna()
if joined.empty: if joined.empty:
return {"overlap_hours": 0} return {"overlap_hours": 0}
g = joined["gauge"].rolling(window_h, min_periods=window_h).sum() g = joined["gauge"].rolling(window_h, min_periods=window_h).sum()
+7 -5
View File
@@ -182,14 +182,18 @@ def _model_forecast(
) )
p_warning = _sigmoid_probability(predicted_max, warn_thr, sigma_h) p_warning = _sigmoid_probability(predicted_max, warn_thr, sigma_h)
if warn_head is not None: if warn_head is not None:
p_warning = max(p_warning, float(warn_head.predict_proba(feature_row)[0][1])) p_warning = max(
p_warning, float(warn_head.predict_proba(feature_row)[0][1])
)
danger_head = ( danger_head = (
None if thresholds_stale else bundle["heads"].get(f"danger_{horizon_h}") None if thresholds_stale else bundle["heads"].get(f"danger_{horizon_h}")
) )
p_danger = _sigmoid_probability(predicted_max, danger_thr, sigma_h) p_danger = _sigmoid_probability(predicted_max, danger_thr, sigma_h)
if danger_head is not None: if danger_head is not None:
p_danger = max(p_danger, float(danger_head.predict_proba(feature_row)[0][1])) p_danger = max(
p_danger, float(danger_head.predict_proba(feature_row)[0][1])
)
p_warning = _clip_probability(p_warning) p_warning = _clip_probability(p_warning)
p_danger = min(_clip_probability(p_danger), p_warning) p_danger = min(_clip_probability(p_danger), p_warning)
@@ -400,6 +404,4 @@ def get_latest_forecasts(
logger.warning("dam state unavailable; dam features will be NaN") logger.warning("dam state unavailable; dam features will be NaN")
dam = pd.DataFrame() dam = pd.DataFrame()
return get_forecasts( return get_forecasts(readings_by_station, models_dir=models_dir, rain=rain, dam=dam)
readings_by_station, models_dir=models_dir, rain=rain, dam=dam
)
+4 -9
View File
@@ -121,12 +121,8 @@ def load_history(
cursor = fetch_from.date() cursor = fetch_from.date()
try: try:
while cursor <= end: while cursor <= end:
chunk_end = min( chunk_end = min(datetime.date(cursor.year, 12, 31), end)
datetime.date(cursor.year, 12, 31), end chunks.append(fetch_history(cursor.isoformat(), chunk_end.isoformat()))
)
chunks.append(
fetch_history(cursor.isoformat(), chunk_end.isoformat())
)
cursor = datetime.date(cursor.year + 1, 1, 1) cursor = datetime.date(cursor.year + 1, 1, 1)
except Exception as error: except Exception as error:
logger.warning(f"Open-Meteo history fetch failed: {error}") logger.warning(f"Open-Meteo history fetch failed: {error}")
@@ -174,7 +170,7 @@ def backfill_db(engine, db_type: str, chunk_rows: int = 5000) -> int:
return 0 return 0
total = 0 total = 0
for start in range(0, len(history), chunk_rows): for start in range(0, len(history), chunk_rows):
part = history.iloc[start: start + chunk_rows] part = history.iloc[start : start + chunk_rows]
total += save_to_db(part, engine, db_type) total += save_to_db(part, engine, db_type)
logger.info(f"openmeteo_rain backfill: {total}/{len(history)} rows") logger.info(f"openmeteo_rain backfill: {total}/{len(history)} rows")
return total return total
@@ -204,8 +200,7 @@ def save_to_db(df: pd.DataFrame, engine, db_type: str) -> int:
cols = ["timestamp"] + point_cols + ["catchment_mean"] cols = ["timestamp"] + point_cols + ["catchment_mean"]
placeholders = ", ".join(f":{c}" for c in cols) placeholders = ", ".join(f":{c}" for c in cols)
updates = ", ".join( updates = ", ".join(
f"{c} = " f"{c} = " + (f"VALUES({c})" if db_type == "mysql" else f"EXCLUDED.{c}")
+ (f"VALUES({c})" if db_type == "mysql" else f"EXCLUDED.{c}")
for c in cols[1:] for c in cols[1:]
) )
if db_type == "mysql": if db_type == "mysql":
+2 -3
View File
@@ -53,6 +53,7 @@ class RainUnavailableError(RuntimeError):
overwrite the deployed v3 artifacts without anyone noticing. overwrite the deployed v3 artifacts without anyone noticing.
""" """
HGB_PARAMS = { HGB_PARAMS = {
"max_iter": 300, "max_iter": 300,
"learning_rate": 0.06, "learning_rate": 0.06,
@@ -518,9 +519,7 @@ def train_all(
"training v3-style bundles WITHOUT dam features" "training v3-style bundles WITHOUT dam features"
) )
if dam_frame is not None: if dam_frame is not None:
logger.info( logger.info(f"dam series: {dam_frame.index.min()} .. {dam_frame.index.max()}")
f"dam series: {dam_frame.index.min()} .. {dam_frame.index.max()}"
)
# Run-level version: v4 only if some requested station actually receives # Run-level version: v4 only if some requested station actually receives
# dam columns (they are gated to DAM_STATIONS; per-bundle versions are # dam columns (they are gated to DAM_STATIONS; per-bundle versions are
+2 -6
View File
@@ -402,9 +402,7 @@ class RidReservoirStore:
measure_row = { measure_row = {
c: _bounded(record.get(c), _MEASURE_BOUNDS[c]) for c in measure_cols c: _bounded(record.get(c), _MEASURE_BOUNDS[c]) for c in measure_cols
} }
measure_row.update( measure_row.update({"dam_id": record["dam_id"], "date": record["date"]})
{"dam_id": record["dam_id"], "date": record["date"]}
)
measurements.append(measure_row) measurements.append(measure_row)
try: try:
with self.engine.begin() as conn: with self.engine.begin() as conn:
@@ -495,9 +493,7 @@ def backfill(
if not store.engine and not store.connect(): if not store.engine and not store.connect():
logger.error("backfill aborted: database connection failed") logger.error("backfill aborted: database connection failed")
return 0 return 0
span = [ span = [start + datetime.timedelta(days=i) for i in range((end - start).days + 1)]
start + datetime.timedelta(days=i) for i in range((end - start).days + 1)
]
present = store.present_dates(start, end) present = store.present_dates(start, end)
targets = [d for d in span if d not in present] targets = [d for d in span if d not in present]
logger.info( logger.info(
+1 -3
View File
@@ -632,9 +632,7 @@ class EnhancedWaterMonitorScraper:
if data: if data:
if self.save_to_database(data): if self.save_to_database(data):
filled_count += len(data) filled_count += len(data)
logger.info( logger.info(f"Filled {len(data)} measurements for {fetch_date}")
f"Filled {len(data)} measurements for {fetch_date}"
)
else: else:
logger.warning(f"Failed to save data for {fetch_date}") logger.warning(f"Failed to save data for {fetch_date}")
else: else:
+17 -16
View File
@@ -374,9 +374,7 @@ async def background_scraping_task():
hii_counts = await asyncio.get_event_loop().run_in_executor( hii_counts = await asyncio.get_event_loop().run_in_executor(
None, hii_collector.run_cycle None, hii_collector.run_cycle
) )
set_gauge( set_gauge("hii_rainfall_rows_saved", hii_counts["rainfall"])
"hii_rainfall_rows_saved", hii_counts["rainfall"]
)
set_gauge( set_gauge(
"hii_waterlevel_rows_saved", hii_counts["waterlevel"] "hii_waterlevel_rows_saved", hii_counts["waterlevel"]
) )
@@ -521,7 +519,9 @@ _STATIC_DIR = os.path.dirname(_DASHBOARD_HTML_PATH)
@app.get("/robots.txt", include_in_schema=False) @app.get("/robots.txt", include_in_schema=False)
async def robots_txt(): async def robots_txt():
return FileResponse(os.path.join(_STATIC_DIR, "robots.txt"), media_type="text/plain") return FileResponse(
os.path.join(_STATIC_DIR, "robots.txt"), media_type="text/plain"
)
@app.get("/llms.txt", include_in_schema=False) @app.get("/llms.txt", include_in_schema=False)
@@ -1022,8 +1022,12 @@ async def get_hii_rainfall_catchment(
engine = _hii_engine() engine = _hii_engine()
if engine is None: if engine is None:
return {"box": hii_rain.CATCHMENT_BOX, "gauge": [], "openmeteo": [], return {
"comparison_24h_sums": {"overlap_hours": 0}} "box": hii_rain.CATCHMENT_BOX,
"gauge": [],
"openmeteo": [],
"comparison_24h_sums": {"overlap_hours": 0},
}
gauge = hii_rain.load_gauge_mean(start=pd.Timestamp(start), engine=engine) gauge = hii_rain.load_gauge_mean(start=pd.Timestamp(start), engine=engine)
openmeteo = None openmeteo = None
try: try:
@@ -1050,7 +1054,10 @@ async def get_hii_rainfall_catchment(
if s is None: if s is None:
return [] return []
return [ return [
{"timestamp": ts.isoformat(), "rain_mm": None if pd.isna(v) else round(float(v), 2)} {
"timestamp": ts.isoformat(),
"rain_mm": None if pd.isna(v) else round(float(v), 2),
}
for ts, v in s.items() for ts, v in s.items()
] ]
@@ -1121,9 +1128,7 @@ async def get_postgres_history(
return cached[1] return cached[1]
try: try:
db_config = Config.get_database_config() db_config = Config.get_database_config()
end_time = ( end_time = datetime.combine(end, datetime.max.time()) if end else datetime.now()
datetime.combine(end, datetime.max.time()) if end else datetime.now()
)
start_time = ( start_time = (
datetime.combine(start, datetime.min.time()) datetime.combine(start, datetime.min.time())
if start if start
@@ -1216,17 +1221,13 @@ async def get_forecast_history(
store = app_state.get("forecast_store") store = app_state.get("forecast_store")
if not store: if not store:
return [] return []
end_dt = ( end_dt = datetime.combine(end, datetime.max.time()) if end else datetime.now()
datetime.combine(end, datetime.max.time()) if end else datetime.now()
)
start_dt = ( start_dt = (
datetime.combine(start, datetime.min.time()) datetime.combine(start, datetime.min.time())
if start if start
else end_dt - timedelta(hours=hours) else end_dt - timedelta(hours=hours)
) )
return await asyncio.to_thread( return await asyncio.to_thread(store.fetch, station_code, start_dt, end_dt, horizon)
store.fetch, station_code, start_dt, end_dt, horizon
)
@app.get("/measurements/latest", response_model=List[MeasurementResponse]) @app.get("/measurements/latest", response_model=List[MeasurementResponse])