Files

503 lines
18 KiB
Python

"""Generate a preliminary CycloneDX inventory from the local Python and Flutter locks."""
from __future__ import annotations
import argparse
import gzip
import hashlib
import importlib.metadata
import importlib.util
import json
import os
import re
from datetime import UTC, datetime
from pathlib import Path
from urllib.parse import unquote, urlparse
from uuid import uuid4
EASYOCR_RUNTIME_MODELS = {
"arabic.pth": {
"url": "https://github.com/JaidedAI/EasyOCR/releases/download/pre-v1.1.6/arabic.zip",
"md5": "993074555550e4e06a6077d55ff0449a",
},
"english_g2.pth": {
"url": "https://github.com/JaidedAI/EasyOCR/releases/download/v1.3/english_g2.zip",
"md5": "5864788e1821be9e454ec108d61b887d",
},
"craft_mlt_25k.pth": {
"url": "https://github.com/JaidedAI/EasyOCR/releases/download/pre-v1.1.6/craft_mlt_25k.zip",
"md5": "2f8227d2def4037cdb3b34389dcf9ec1",
},
}
def file_hash(path: Path, algorithm: str = "sha256") -> str:
hasher = (
hashlib.md5(usedforsecurity=False)
if algorithm == "md5"
else hashlib.new(algorithm)
)
with path.open("rb") as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
hasher.update(chunk)
return hasher.hexdigest()
def digest(path: Path) -> str:
return file_hash(path)
def file_component(
path: Path,
*,
name: str,
component_type: str = "file",
scope: str,
license_status: str,
evidence: dict[str, str] | None = None,
) -> dict[str, object]:
sha256 = digest(path)
properties = [
{"name": "inventory.scope", "value": scope},
{"name": "inventory.observed_filename", "value": path.name},
{"name": "inventory.size_bytes", "value": str(path.stat().st_size)},
{"name": "inventory.license_status", "value": license_status},
]
properties.extend(
{"name": f"inventory.evidence.{key}", "value": value}
for key, value in (evidence or {}).items()
)
return {
"type": component_type,
"bom-ref": f"file:{name}:sha256:{sha256}",
"name": name,
"hashes": [{"alg": "SHA-256", "content": sha256}],
"properties": properties,
}
def _ocr_model_directory() -> Path:
configured = os.getenv("LOCAL_OCR_MODEL_DIR", "").strip()
if configured:
return Path(configured).expanduser()
local_app_data = os.getenv("LOCALAPPDATA")
base = Path(local_app_data) if local_app_data else Path.home() / ".local" / "share"
return base / "SovereignAI" / "models" / "easyocr"
def ocr_runtime_components(model_dir: Path) -> tuple[list[dict[str, object]], list[str]]:
"""Record installed EasyOCR weights and report requested model files that are absent."""
components: list[dict[str, object]] = []
missing: list[str] = []
for filename, source in EASYOCR_RUNTIME_MODELS.items():
path = model_dir / filename
if not path.is_file():
missing.append(filename)
continue
md5 = file_hash(path, "md5") # upstream manifest uses MD5 for artifact identity only
components.append(
file_component(
path,
name=f"EasyOCR model {filename}",
component_type="machine-learning-model",
scope="local-runtime-artifact",
license_status="model redistribution terms not established; human review required",
evidence={
"upstream_manifest": "easyocr==1.7.2 config.py",
"upstream_url": source["url"],
"upstream_manifest_md5": source["md5"],
"observed_md5": md5,
"upstream_md5_match": str(md5 == source["md5"]).lower(),
},
)
)
return components, missing
def pdfium_runtime_components() -> list[dict[str, object]]:
"""Hash the installed PDFium native binary and its wheel-bundled notice evidence."""
try:
spec = importlib.util.find_spec("pypdfium2_raw")
except (ImportError, ValueError):
return []
if spec is None or not spec.submodule_search_locations:
return []
package_root = Path(next(iter(spec.submodule_search_locations)))
binaries = [package_root / name for name in ("pdfium.dll", "libpdfium.so", "libpdfium.dylib")]
components: list[dict[str, object]] = []
try:
distribution = importlib.metadata.distribution("pypdfium2")
except importlib.metadata.PackageNotFoundError:
distribution = None
notice_hashes: dict[str, str] = {}
if distribution is not None:
for item in distribution.files or ():
relative = str(item).replace("\\", "/")
if "/BUILD_LICENSES/" not in relative:
continue
notice = Path(distribution.locate_file(item))
if notice.is_file():
notice_hashes[relative] = digest(notice)
for binary in binaries:
if binary.is_file():
components.append(
file_component(
binary,
name=f"PDFium native binary {binary.name}",
scope="local-runtime-artifact",
license_status="upstream and bundled dependency notices recorded; release review required",
evidence={
"python_distribution": (
f"pypdfium2 {distribution.version}"
if distribution is not None
else "pypdfium2 distribution metadata unavailable"
),
"bundled_build_license_files": json.dumps(
notice_hashes, ensure_ascii=False, sort_keys=True
),
},
)
)
return components
def canonical_name(value: str) -> str:
return re.sub(r"[-_.]+", "-", value).lower()
def requirement_names(path: Path) -> set[str]:
names: set[str] = set()
for raw_line in path.read_text(encoding="utf-8").splitlines():
line = raw_line.split("#", 1)[0].strip()
match = re.match(r"([A-Za-z0-9_.-]+)", line)
if match:
names.add(canonical_name(match.group(1)))
return names
def python_components(root: Path) -> list[dict[str, object]]:
direct = requirement_names(root / "requirements.txt")
optional = requirement_names(root / "requirements-ocr.txt")
components: list[dict[str, object]] = []
seen: set[str] = set()
for distribution in importlib.metadata.distributions():
name = distribution.metadata.get("Name")
version = distribution.version
if not name or not version:
continue
key = canonical_name(name)
if key in seen:
continue
seen.add(key)
raw_license = (
distribution.metadata.get("License-Expression")
or distribution.metadata.get("License")
or ""
).strip()
if not raw_license:
raw_license = next(
(
value.removeprefix("License :: ")
for value in distribution.metadata.get_all("Classifier", [])
if value.startswith("License :: ")
),
"",
)
scope = (
"optional-ocr"
if key in optional
else "direct-runtime"
if key in direct
else "installed-transitive-or-development"
)
components.append(
component(
ecosystem="pypi",
name=name,
version=version,
scope=scope,
raw_license=raw_license,
license_source="installed Python distribution metadata",
)
)
return sorted(components, key=lambda item: str(item["name"]).lower())
def component(
*,
ecosystem: str,
name: str,
version: str,
scope: str,
raw_license: str,
license_source: str,
license_file: Path | None = None,
) -> dict[str, object]:
normalized = canonical_name(name) if ecosystem == "pypi" else name.lower()
result: dict[str, object] = {
"type": "library",
"bom-ref": f"pkg:{ecosystem}/{normalized}@{version}",
"name": name,
"version": version,
"purl": f"pkg:{ecosystem}/{normalized}@{version}",
"properties": [
{"name": "inventory.scope", "value": scope},
{"name": "inventory.license_source", "value": license_source},
],
}
if raw_license:
result["licenses"] = [{"license": {"name": raw_license}}]
else:
result["properties"].append(
{"name": "inventory.license_status", "value": "not-present-in-metadata"}
)
if license_file is not None:
result["properties"].extend(
[
{
"name": "inventory.license_file",
"value": license_file.name,
},
{
"name": "inventory.license_file_sha256",
"value": digest(license_file),
},
]
)
return result
def parse_lockfile(path: Path) -> list[dict[str, str]]:
packages: list[dict[str, str]] = []
current: dict[str, str] | None = None
package_header = re.compile(r"^ ([A-Za-z0-9_]+):$")
for line in path.read_text(encoding="utf-8").splitlines():
match = package_header.match(line)
if match:
if current is not None and current.get("version"):
packages.append(current)
current = {"name": match.group(1)}
continue
if current is None:
continue
for field in ("dependency", "source", "version"):
field_match = re.match(rf"^ {field}:\s*(.*?)\s*$", line)
if field_match:
current[field] = field_match.group(1).strip('"\'')
if current is not None and current.get("version"):
packages.append(current)
return packages
def package_roots(config_path: Path) -> dict[str, Path]:
config = json.loads(config_path.read_text(encoding="utf-8"))
roots: dict[str, Path] = {}
for package in config.get("packages", []):
uri = urlparse(package.get("rootUri", ""))
if uri.scheme == "file":
raw_path = unquote(uri.path)
if re.match(r"^/[A-Za-z]:/", raw_path):
raw_path = raw_path[1:]
roots[package["name"]] = Path(raw_path)
return roots
def flutter_license_file(package_root: Path) -> Path | None:
try:
files = {item.name.lower(): item for item in package_root.iterdir() if item.is_file()}
except OSError:
return None
for name in ("license", "license.txt", "license.md", "copying", "copying.txt"):
if name in files:
return files[name]
return None
def flutter_components(root: Path) -> list[dict[str, object]]:
app_root = root / "flutter_app"
packages = parse_lockfile(app_root / "pubspec.lock")
roots = package_roots(app_root / ".dart_tool" / "package_config.json")
components: list[dict[str, object]] = []
for package in packages:
name = package["name"]
version = package["version"]
scope = package.get("dependency", "transitive")
package_root = roots.get(name)
license_file = flutter_license_file(package_root) if package_root else None
declared = ""
license_source = "resolved pubspec.lock; license declaration unavailable"
if package.get("source") == "sdk" and package_root is not None:
sdk_license = package_root.parent.parent / "LICENSE"
if sdk_license.is_file():
license_file = sdk_license
declared = "Flutter SDK root license notice (BSD-style)"
license_source = "Flutter SDK root LICENSE inherited by SDK package"
if package_root:
pubspec = package_root / "pubspec.yaml"
if pubspec.exists() and not declared:
match = re.search(
r"(?m)^license:\s*(.*?)\s*$",
pubspec.read_text(encoding="utf-8", errors="replace"),
)
if match:
declared = match.group(1).strip('"\'')
license_source = "package pubspec.yaml declaration"
if not declared and license_file:
declared = f"See included {license_file.name} file; not automatically classified"
license_source = "package license file (hash recorded; human classification required)"
components.append(
component(
ecosystem="pub",
name=name,
version=version,
scope=scope,
raw_license=declared,
license_source=license_source,
license_file=license_file,
)
)
return sorted(components, key=lambda item: str(item["name"]).lower())
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--project-root",
type=Path,
default=Path(__file__).resolve().parents[1],
)
parser.add_argument(
"--output",
type=Path,
default=None,
help="Output JSON path (default: sbom/component-inventory.json)",
)
parser.add_argument(
"--flutter-notices",
type=Path,
default=None,
help="Optional Flutter build's bundled NOTICES.Z archive for evidence/hash",
)
args = parser.parse_args()
root = args.project_root.resolve()
output = args.output or root / "sbom" / "component-inventory.json"
ocr_components, missing_ocr_models = ocr_runtime_components(_ocr_model_directory())
pdfium_components = pdfium_runtime_components()
components = (
python_components(root)
+ flutter_components(root)
+ pdfium_components
+ ocr_components
)
metadata: dict[str, object] = {
"timestamp": datetime.now(UTC).isoformat(),
"tools": [{"name": "generate_component_inventory.py"}],
"component": {
"type": "application",
"name": "SovereignAI-Starter",
"version": "1.0.0",
},
"properties": [
{
"name": "inventory.note",
"value": "Preliminary local environment inventory; not a legal approval or a complete commercial SBOM.",
},
{
"name": "inventory.python_requirements_sha256",
"value": digest(root / "requirements.txt"),
},
{
"name": "inventory.ocr_requirements_sha256",
"value": digest(root / "requirements-ocr.txt"),
},
{
"name": "inventory.flutter_lock_sha256",
"value": digest(root / "flutter_app" / "pubspec.lock"),
},
{
"name": "inventory.runtime_artifact_scope",
"value": "Observed files from the current host only; not a release artifact manifest.",
},
],
}
metadata["properties"].extend(
{"name": "inventory.missing_ocr_model", "value": filename}
for filename in missing_ocr_models
)
if args.flutter_notices:
notice_bytes = args.flutter_notices.read_bytes()
metadata["properties"].extend(
[
{
"name": "inventory.flutter_notices_sha256",
"value": hashlib.sha256(notice_bytes).hexdigest(),
},
{
"name": "inventory.flutter_notices_uncompressed_bytes",
"value": str(len(gzip.decompress(notice_bytes))),
},
]
)
document = {
"bomFormat": "CycloneDX",
"specVersion": "1.5",
"serialNumber": f"urn:uuid:{uuid4()}",
"version": 1,
"metadata": metadata,
"components": components,
}
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(
json.dumps(document, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
python_count = sum(str(item.get("purl", "")).startswith("pkg:pypi/") for item in components)
pub_count = sum(str(item.get("purl", "")).startswith("pkg:pub/") for item in components)
missing_library_license_metadata = sum(
item.get("type") == "library" and "licenses" not in item
for item in components
)
runtime_license_review = sum(
item.get("type") in {"file", "machine-learning-model"}
and any(
prop["name"] == "inventory.license_status"
and "review required" in prop["value"]
for prop in item["properties"]
)
for item in components
)
unclassified = sum(
any(
prop["name"] == "inventory.license_source"
and "human classification required" in prop["value"]
for prop in item["properties"]
)
for item in components
)
hashed_licenses = sum(
any(prop["name"] == "inventory.license_file_sha256" for prop in item["properties"])
for item in components
)
print(
json.dumps(
{
"output": str(output),
"python_components": python_count,
"flutter_components": pub_count,
"pdfium_runtime_artifacts": len(pdfium_components),
"ocr_runtime_artifacts": len(ocr_components),
"missing_ocr_models": missing_ocr_models,
"library_components_missing_license_metadata": missing_library_license_metadata,
"runtime_artifacts_requiring_license_review": runtime_license_review,
"license_files_hashed": hashed_licenses,
"license_files_requiring_manual_classification": unclassified,
"complete_commercial_audit": False,
},
ensure_ascii=False,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(main())