"""Generate a preliminary CycloneDX inventory from the local Python and Flutter locks.""" from __future__ import annotations import argparse import gzip import hashlib import importlib.metadata import importlib.util import json import os import re from datetime import UTC, datetime from pathlib import Path from urllib.parse import unquote, urlparse from uuid import uuid4 EASYOCR_RUNTIME_MODELS = { "arabic.pth": { "url": "https://github.com/JaidedAI/EasyOCR/releases/download/pre-v1.1.6/arabic.zip", "md5": "993074555550e4e06a6077d55ff0449a", }, "english_g2.pth": { "url": "https://github.com/JaidedAI/EasyOCR/releases/download/v1.3/english_g2.zip", "md5": "5864788e1821be9e454ec108d61b887d", }, "craft_mlt_25k.pth": { "url": "https://github.com/JaidedAI/EasyOCR/releases/download/pre-v1.1.6/craft_mlt_25k.zip", "md5": "2f8227d2def4037cdb3b34389dcf9ec1", }, } def file_hash(path: Path, algorithm: str = "sha256") -> str: hasher = ( hashlib.md5(usedforsecurity=False) if algorithm == "md5" else hashlib.new(algorithm) ) with path.open("rb") as stream: for chunk in iter(lambda: stream.read(1024 * 1024), b""): hasher.update(chunk) return hasher.hexdigest() def digest(path: Path) -> str: return file_hash(path) def file_component( path: Path, *, name: str, component_type: str = "file", scope: str, license_status: str, evidence: dict[str, str] | None = None, ) -> dict[str, object]: sha256 = digest(path) properties = [ {"name": "inventory.scope", "value": scope}, {"name": "inventory.observed_filename", "value": path.name}, {"name": "inventory.size_bytes", "value": str(path.stat().st_size)}, {"name": "inventory.license_status", "value": license_status}, ] properties.extend( {"name": f"inventory.evidence.{key}", "value": value} for key, value in (evidence or {}).items() ) return { "type": component_type, "bom-ref": f"file:{name}:sha256:{sha256}", "name": name, "hashes": [{"alg": "SHA-256", "content": sha256}], "properties": properties, } def _ocr_model_directory() -> Path: configured = os.getenv("LOCAL_OCR_MODEL_DIR", "").strip() if configured: return Path(configured).expanduser() local_app_data = os.getenv("LOCALAPPDATA") base = Path(local_app_data) if local_app_data else Path.home() / ".local" / "share" return base / "SovereignAI" / "models" / "easyocr" def ocr_runtime_components(model_dir: Path) -> tuple[list[dict[str, object]], list[str]]: """Record installed EasyOCR weights and report requested model files that are absent.""" components: list[dict[str, object]] = [] missing: list[str] = [] for filename, source in EASYOCR_RUNTIME_MODELS.items(): path = model_dir / filename if not path.is_file(): missing.append(filename) continue md5 = file_hash(path, "md5") # upstream manifest uses MD5 for artifact identity only components.append( file_component( path, name=f"EasyOCR model {filename}", component_type="machine-learning-model", scope="local-runtime-artifact", license_status="model redistribution terms not established; human review required", evidence={ "upstream_manifest": "easyocr==1.7.2 config.py", "upstream_url": source["url"], "upstream_manifest_md5": source["md5"], "observed_md5": md5, "upstream_md5_match": str(md5 == source["md5"]).lower(), }, ) ) return components, missing def pdfium_runtime_components() -> list[dict[str, object]]: """Hash the installed PDFium native binary and its wheel-bundled notice evidence.""" try: spec = importlib.util.find_spec("pypdfium2_raw") except (ImportError, ValueError): return [] if spec is None or not spec.submodule_search_locations: return [] package_root = Path(next(iter(spec.submodule_search_locations))) binaries = [package_root / name for name in ("pdfium.dll", "libpdfium.so", "libpdfium.dylib")] components: list[dict[str, object]] = [] try: distribution = importlib.metadata.distribution("pypdfium2") except importlib.metadata.PackageNotFoundError: distribution = None notice_hashes: dict[str, str] = {} if distribution is not None: for item in distribution.files or (): relative = str(item).replace("\\", "/") if "/BUILD_LICENSES/" not in relative: continue notice = Path(distribution.locate_file(item)) if notice.is_file(): notice_hashes[relative] = digest(notice) for binary in binaries: if binary.is_file(): components.append( file_component( binary, name=f"PDFium native binary {binary.name}", scope="local-runtime-artifact", license_status="upstream and bundled dependency notices recorded; release review required", evidence={ "python_distribution": ( f"pypdfium2 {distribution.version}" if distribution is not None else "pypdfium2 distribution metadata unavailable" ), "bundled_build_license_files": json.dumps( notice_hashes, ensure_ascii=False, sort_keys=True ), }, ) ) return components def canonical_name(value: str) -> str: return re.sub(r"[-_.]+", "-", value).lower() def requirement_names(path: Path) -> set[str]: names: set[str] = set() for raw_line in path.read_text(encoding="utf-8").splitlines(): line = raw_line.split("#", 1)[0].strip() match = re.match(r"([A-Za-z0-9_.-]+)", line) if match: names.add(canonical_name(match.group(1))) return names def python_components(root: Path) -> list[dict[str, object]]: direct = requirement_names(root / "requirements.txt") optional = requirement_names(root / "requirements-ocr.txt") components: list[dict[str, object]] = [] seen: set[str] = set() for distribution in importlib.metadata.distributions(): name = distribution.metadata.get("Name") version = distribution.version if not name or not version: continue key = canonical_name(name) if key in seen: continue seen.add(key) raw_license = ( distribution.metadata.get("License-Expression") or distribution.metadata.get("License") or "" ).strip() if not raw_license: raw_license = next( ( value.removeprefix("License :: ") for value in distribution.metadata.get_all("Classifier", []) if value.startswith("License :: ") ), "", ) scope = ( "optional-ocr" if key in optional else "direct-runtime" if key in direct else "installed-transitive-or-development" ) components.append( component( ecosystem="pypi", name=name, version=version, scope=scope, raw_license=raw_license, license_source="installed Python distribution metadata", ) ) return sorted(components, key=lambda item: str(item["name"]).lower()) def component( *, ecosystem: str, name: str, version: str, scope: str, raw_license: str, license_source: str, license_file: Path | None = None, ) -> dict[str, object]: normalized = canonical_name(name) if ecosystem == "pypi" else name.lower() result: dict[str, object] = { "type": "library", "bom-ref": f"pkg:{ecosystem}/{normalized}@{version}", "name": name, "version": version, "purl": f"pkg:{ecosystem}/{normalized}@{version}", "properties": [ {"name": "inventory.scope", "value": scope}, {"name": "inventory.license_source", "value": license_source}, ], } if raw_license: result["licenses"] = [{"license": {"name": raw_license}}] else: result["properties"].append( {"name": "inventory.license_status", "value": "not-present-in-metadata"} ) if license_file is not None: result["properties"].extend( [ { "name": "inventory.license_file", "value": license_file.name, }, { "name": "inventory.license_file_sha256", "value": digest(license_file), }, ] ) return result def parse_lockfile(path: Path) -> list[dict[str, str]]: packages: list[dict[str, str]] = [] current: dict[str, str] | None = None package_header = re.compile(r"^ ([A-Za-z0-9_]+):$") for line in path.read_text(encoding="utf-8").splitlines(): match = package_header.match(line) if match: if current is not None and current.get("version"): packages.append(current) current = {"name": match.group(1)} continue if current is None: continue for field in ("dependency", "source", "version"): field_match = re.match(rf"^ {field}:\s*(.*?)\s*$", line) if field_match: current[field] = field_match.group(1).strip('"\'') if current is not None and current.get("version"): packages.append(current) return packages def package_roots(config_path: Path) -> dict[str, Path]: config = json.loads(config_path.read_text(encoding="utf-8")) roots: dict[str, Path] = {} for package in config.get("packages", []): uri = urlparse(package.get("rootUri", "")) if uri.scheme == "file": raw_path = unquote(uri.path) if re.match(r"^/[A-Za-z]:/", raw_path): raw_path = raw_path[1:] roots[package["name"]] = Path(raw_path) return roots def flutter_license_file(package_root: Path) -> Path | None: try: files = {item.name.lower(): item for item in package_root.iterdir() if item.is_file()} except OSError: return None for name in ("license", "license.txt", "license.md", "copying", "copying.txt"): if name in files: return files[name] return None def flutter_components(root: Path) -> list[dict[str, object]]: app_root = root / "flutter_app" packages = parse_lockfile(app_root / "pubspec.lock") roots = package_roots(app_root / ".dart_tool" / "package_config.json") components: list[dict[str, object]] = [] for package in packages: name = package["name"] version = package["version"] scope = package.get("dependency", "transitive") package_root = roots.get(name) license_file = flutter_license_file(package_root) if package_root else None declared = "" license_source = "resolved pubspec.lock; license declaration unavailable" if package.get("source") == "sdk" and package_root is not None: sdk_license = package_root.parent.parent / "LICENSE" if sdk_license.is_file(): license_file = sdk_license declared = "Flutter SDK root license notice (BSD-style)" license_source = "Flutter SDK root LICENSE inherited by SDK package" if package_root: pubspec = package_root / "pubspec.yaml" if pubspec.exists() and not declared: match = re.search( r"(?m)^license:\s*(.*?)\s*$", pubspec.read_text(encoding="utf-8", errors="replace"), ) if match: declared = match.group(1).strip('"\'') license_source = "package pubspec.yaml declaration" if not declared and license_file: declared = f"See included {license_file.name} file; not automatically classified" license_source = "package license file (hash recorded; human classification required)" components.append( component( ecosystem="pub", name=name, version=version, scope=scope, raw_license=declared, license_source=license_source, license_file=license_file, ) ) return sorted(components, key=lambda item: str(item["name"]).lower()) def main() -> int: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument( "--project-root", type=Path, default=Path(__file__).resolve().parents[1], ) parser.add_argument( "--output", type=Path, default=None, help="Output JSON path (default: sbom/component-inventory.json)", ) parser.add_argument( "--flutter-notices", type=Path, default=None, help="Optional Flutter build's bundled NOTICES.Z archive for evidence/hash", ) args = parser.parse_args() root = args.project_root.resolve() output = args.output or root / "sbom" / "component-inventory.json" ocr_components, missing_ocr_models = ocr_runtime_components(_ocr_model_directory()) pdfium_components = pdfium_runtime_components() components = ( python_components(root) + flutter_components(root) + pdfium_components + ocr_components ) metadata: dict[str, object] = { "timestamp": datetime.now(UTC).isoformat(), "tools": [{"name": "generate_component_inventory.py"}], "component": { "type": "application", "name": "SovereignAI-Starter", "version": "1.0.0", }, "properties": [ { "name": "inventory.note", "value": "Preliminary local environment inventory; not a legal approval or a complete commercial SBOM.", }, { "name": "inventory.python_requirements_sha256", "value": digest(root / "requirements.txt"), }, { "name": "inventory.ocr_requirements_sha256", "value": digest(root / "requirements-ocr.txt"), }, { "name": "inventory.flutter_lock_sha256", "value": digest(root / "flutter_app" / "pubspec.lock"), }, { "name": "inventory.runtime_artifact_scope", "value": "Observed files from the current host only; not a release artifact manifest.", }, ], } metadata["properties"].extend( {"name": "inventory.missing_ocr_model", "value": filename} for filename in missing_ocr_models ) if args.flutter_notices: notice_bytes = args.flutter_notices.read_bytes() metadata["properties"].extend( [ { "name": "inventory.flutter_notices_sha256", "value": hashlib.sha256(notice_bytes).hexdigest(), }, { "name": "inventory.flutter_notices_uncompressed_bytes", "value": str(len(gzip.decompress(notice_bytes))), }, ] ) document = { "bomFormat": "CycloneDX", "specVersion": "1.5", "serialNumber": f"urn:uuid:{uuid4()}", "version": 1, "metadata": metadata, "components": components, } output.parent.mkdir(parents=True, exist_ok=True) output.write_text( json.dumps(document, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" ) python_count = sum(str(item.get("purl", "")).startswith("pkg:pypi/") for item in components) pub_count = sum(str(item.get("purl", "")).startswith("pkg:pub/") for item in components) missing_library_license_metadata = sum( item.get("type") == "library" and "licenses" not in item for item in components ) runtime_license_review = sum( item.get("type") in {"file", "machine-learning-model"} and any( prop["name"] == "inventory.license_status" and "review required" in prop["value"] for prop in item["properties"] ) for item in components ) unclassified = sum( any( prop["name"] == "inventory.license_source" and "human classification required" in prop["value"] for prop in item["properties"] ) for item in components ) hashed_licenses = sum( any(prop["name"] == "inventory.license_file_sha256" for prop in item["properties"]) for item in components ) print( json.dumps( { "output": str(output), "python_components": python_count, "flutter_components": pub_count, "pdfium_runtime_artifacts": len(pdfium_components), "ocr_runtime_artifacts": len(ocr_components), "missing_ocr_models": missing_ocr_models, "library_components_missing_license_metadata": missing_library_license_metadata, "runtime_artifacts_requiring_license_review": runtime_license_review, "license_files_hashed": hashed_licenses, "license_files_requiring_manual_classification": unclassified, "complete_commercial_audit": False, }, ensure_ascii=False, ) ) return 0 if __name__ == "__main__": raise SystemExit(main())