"""Validated deployment configuration and URL validation for Confluence Web.""" from __future__ import annotations import os import math import posixpath import tempfile import urllib.parse import uuid from pathlib import Path from typing import List, Optional from pydantic import BaseModel, Field, field_validator class Settings(BaseModel): """Explicit validated deployment configuration for backend service.""" # Confluence remote access policy approved_confluence_origins: List[str] = Field( default_factory=lambda: ["https://approved.example.com"] ) corporate_ca_path: Optional[str] = None # Model provider configuration model_provider: str = "fake" # Explicit fake adapter or openai; unknown values fail app creation. model_name: str = "fake-model" model_api_key: Optional[str] = None model_endpoint: Optional[str] = None model_context_window_tokens: int = 128_000 model_max_output_tokens: int = 4096 # Container & Docker configuration runtime_image: str = "confluence-agent:latest" docker_host: Optional[str] = None container_label_key: str = "com.confluence_web.app" container_label_value: str = "query-runner" container_instance_id: str = Field(default_factory=lambda: uuid.uuid4().hex) # Artifact storage artifact_storage_dir: Path = Field( default_factory=lambda: Path(tempfile.gettempdir()) / "confluence_web_artifacts" ) artifact_max_files_per_query: int = 20 artifact_max_bytes_per_file: int = 10 * 1024 * 1024 # 10 MiB artifact_max_bytes_per_query: int = 50 * 1024 * 1024 # 50 MiB artifact_global_storage_limit: int = 500 * 1024 * 1024 # 500 MiB artifact_ttl_seconds: int = 900 # 15 minutes # Server & frontend bind_host: str = "127.0.0.1" bind_port: int = 8000 frontend_dist_dir: Optional[Path] = None # Runtime deadlines and concurrency query_timeout_seconds: float = 180.0 cleanup_timeout_seconds: float = 10.0 max_inflight_remote_calls: int = 4 max_confluence_calls: int = 100 max_model_calls: int = 50 # Development mode (explicit opt-in) dev_mode: bool = False @field_validator("approved_confluence_origins") @classmethod def validate_origins(cls, v: List[str]) -> List[str]: for origin in v: try: parsed = urllib.parse.urlsplit(origin) if parsed.scheme == "http" and parsed.hostname not in ("localhost", "127.0.0.1", "::1"): raise ValueError(f"Approved origin '{origin}' uses HTTP but is not loopback. HTTPS is required.") except Exception as e: if isinstance(e, ValueError): raise raise ValueError(f"Invalid approved origin: {origin}") from e return v @field_validator("query_timeout_seconds", "cleanup_timeout_seconds") @classmethod def validate_deadline(cls, value): if not math.isfinite(value) or value <= 0: raise ValueError("Deadlines must be finite and positive") return value @field_validator("model_context_window_tokens", "model_max_output_tokens", "artifact_ttl_seconds") @classmethod def validate_positive(cls, value): if value <= 0: raise ValueError("Resource limits must be positive") return value @classmethod def from_env(cls) -> Settings: """Load settings from environment variables.""" def _parse_int(name: str, default: int) -> int: val = os.getenv(name) if val is None: return default try: return int(val) except ValueError: raise ValueError(f"Invalid integer environment variable {name}") def _parse_float(name: str, default: float) -> float: val = os.getenv(name) if val is None: return default try: parsed = float(val) if not math.isfinite(parsed): raise ValueError return parsed except ValueError: raise ValueError(f"Invalid float environment variable {name}") origins_raw = os.getenv("CONFLUENCE_WEB_APPROVED_ORIGINS") origins = ( [o.strip() for o in origins_raw.split(",") if o.strip()] if origins_raw else ["https://approved.example.com"] ) storage_dir_raw = os.getenv("CONFLUENCE_WEB_ARTIFACT_DIR") storage_dir = ( Path(storage_dir_raw) if storage_dir_raw else Path(tempfile.gettempdir()) / "confluence_web_artifacts" ) frontend_dist_raw = os.getenv("CONFLUENCE_WEB_FRONTEND_DIST_DIR") frontend_dist = Path(frontend_dist_raw) if frontend_dist_raw else None dev_mode = os.getenv("CONFLUENCE_WEB_DEV_MODE", "false").lower() in ( "1", "true", "yes", ) return cls( approved_confluence_origins=origins, corporate_ca_path=os.getenv("CONFLUENCE_WEB_CORPORATE_CA_PATH"), model_provider=os.getenv("CONFLUENCE_WEB_MODEL_PROVIDER", "fake"), model_name=os.getenv("CONFLUENCE_WEB_MODEL_NAME", "fake-model"), model_api_key=os.getenv("CONFLUENCE_WEB_MODEL_API_KEY"), model_endpoint=os.getenv("CONFLUENCE_WEB_MODEL_ENDPOINT"), model_context_window_tokens=_parse_int( "CONFLUENCE_WEB_MODEL_CONTEXT_WINDOW_TOKENS", 128000 ), model_max_output_tokens=_parse_int( "CONFLUENCE_WEB_MODEL_MAX_OUTPUT_TOKENS", 4096 ), runtime_image=os.getenv( "CONFLUENCE_WEB_RUNTIME_IMAGE", "confluence-agent:latest" ), docker_host=os.getenv("CONFLUENCE_WEB_DOCKER_HOST"), container_label_key=os.getenv( "CONFLUENCE_WEB_CONTAINER_LABEL_KEY", "com.confluence_web.app" ), container_label_value=os.getenv( "CONFLUENCE_WEB_CONTAINER_LABEL_VALUE", "query-runner" ), container_instance_id=os.getenv( "CONFLUENCE_WEB_CONTAINER_INSTANCE_ID", uuid.uuid4().hex ), artifact_storage_dir=storage_dir, bind_host=os.getenv("CONFLUENCE_WEB_BIND_HOST", "127.0.0.1"), bind_port=_parse_int("CONFLUENCE_WEB_BIND_PORT", 8000), frontend_dist_dir=frontend_dist, dev_mode=dev_mode, query_timeout_seconds=_parse_float( "CONFLUENCE_WEB_QUERY_TIMEOUT_SECONDS", 180.0 ), cleanup_timeout_seconds=_parse_float( "CONFLUENCE_WEB_CLEANUP_TIMEOUT_SECONDS", 10.0 ), ) from backend.errors import DestinationDeniedError, InvalidInputError def canonicalize_url(url_str: str) -> str: """Canonicalize a URL: scheme/host/port and context path. Rejects userinfo, fragments, query strings, and path traversal. """ if not isinstance(url_str, str): raise InvalidInputError("URL must be a string") stripped = url_str.strip() if not stripped: raise InvalidInputError("URL cannot be empty") if len(stripped.encode("utf-8")) > 8192: raise InvalidInputError("URL exceeds 8 KiB limit") if any(ord(c) < 32 or ord(c) == 127 for c in stripped) or "\\" in stripped: raise InvalidInputError("URL contains invalid characters") try: parsed = urllib.parse.urlsplit(stripped) except ValueError as exc: raise InvalidInputError("Malformed URL") from exc scheme = parsed.scheme.lower() if scheme not in ("http", "https"): raise InvalidInputError(f"Invalid URL scheme: {scheme}. Only http and https allowed.") if parsed.username or parsed.password or "@" in parsed.netloc: raise InvalidInputError("URL must not contain userinfo") if parsed.fragment or "#" in stripped: raise InvalidInputError("URL must not contain fragments") if parsed.query or "?" in stripped: raise InvalidInputError("URL must not contain query parameters") hostname = parsed.hostname if not hostname: raise InvalidInputError("URL must contain a valid hostname") hostname = hostname.lower() hostname = f"[{hostname}]" if ":" in hostname else hostname # Port canonicalization try: port = parsed.port except ValueError: raise InvalidInputError("URL contains invalid port") if port is not None: if not (1 <= port <= 65535): raise InvalidInputError(f"URL port {port} is out of range") if (scheme == "https" and port == 443) or (scheme == "http" and port == 80): netloc = hostname else: netloc = f"{hostname}:{port}" else: netloc = hostname # Check for traversal attempts in raw path raw_path = parsed.path if "%2e" in raw_path.lower() or "%2f" in raw_path.lower(): raise InvalidInputError("URL contains encoded path traversal") # Reject raw traversal segments like /../ or /./ path_segments = raw_path.split("/") if ".." in path_segments or "." in path_segments: raise InvalidInputError("URL contains path traversal or dot segments") # Normalize path normalized_path = posixpath.normpath(raw_path) if raw_path else "" if normalized_path.startswith("..") or "/../" in normalized_path or normalized_path == "..": raise InvalidInputError("URL contains directory traversal") if normalized_path == "/" or normalized_path == ".": normalized_path = "" elif normalized_path.endswith("/"): normalized_path = normalized_path.rstrip("/") return f"{scheme}://{netloc}{normalized_path}" def validate_confluence_url(user_url: str, approved_origins: List[str]) -> str: """Canonicalize and validate user-supplied Confluence URL against approved origins. Rejects path-prefix lookalikes, non-matching origins, and invalid URLs. Returns canonical approved URL. """ canonical_user = canonicalize_url(user_url) user_parsed = urllib.parse.urlsplit(canonical_user) if user_parsed.scheme == "http" and user_parsed.hostname not in ("localhost", "127.0.0.1", "::1"): raise DestinationDeniedError( f"Confluence destination denied: HTTP is only allowed for loopback development; HTTPS required for '{user_url}'" ) for approved in approved_origins: try: canonical_approved = canonicalize_url(approved) except Exception: continue approved_parsed = urllib.parse.urlsplit(canonical_approved) # Scheme and netloc must match exactly if (user_parsed.scheme, user_parsed.netloc) != (approved_parsed.scheme, approved_parsed.netloc): continue approved_path = approved_parsed.path.rstrip("/") user_path = user_parsed.path.rstrip("/") # Context path matching: must match exactly or be an approved subpath if user_path == approved_path: return canonical_user # If approved has no context path (root), user must match root or subpath if not approved_path and (not user_path or user_path.startswith("/")): return canonical_user # If approved has context path like /wiki, user_path must start with /wiki/ if user_path.startswith(approved_path + "/"): return canonical_user raise DestinationDeniedError(f"Confluence destination denied by policy: '{user_url}' is not in approved origins")