from __future__ import annotations from typing import Any from .config import DATA_DISCLAIMER, GENERATOR_VERSION, GeneratorConfig from .models import DatasetBundle SKILL_IDS: tuple[str, ...] = ( 'ship-project-master-planning', 'ship-wbs-network-planning', 'ship-bom-explosion', 'ship-mrp-netting', 'ship-make-buy-outsource-decision', 'ship-material-readiness', 'ship-block-production-scheduling', 'ship-dock-erection-scheduling', 'ship-crane-lift-scheduling', 'ship-yard-space-scheduling', 'ship-outfitting-zone-scheduling', 'ship-workforce-scheduling', 'ship-procurement-planning', 'ship-outsourcing-planning', 'ship-quality-hold-planning', 'ship-scenario-simulation', 'ship-schedule-repair', 'ship-schedule-explanation', 'ship-bottleneck-detection', 'ship-risk-prediction', ) RAG_CATEGORIES: tuple[str, ...] = ( '\u8239\u8236\u5efa\u9020\u6d41\u7a0b', '\u5206\u6bb5\u5236\u9020\u5de5\u827a\u89c4\u5219', '\u603b\u7ec4\u548c\u642d\u8f7d\u89c4\u5219', '\u8239\u575e\u4f7f\u7528\u89c4\u5219', '\u9f99\u95e8\u540a\u540a\u88c5\u89c4\u5219', '\u5206\u6bb5\u8fd0\u8f93\u89c4\u5219', '\u6d82\u88c5\u73af\u5883\u89c4\u5219', '\u823e\u88c5\u524d\u79fb\u89c4\u5219', '\u6258\u76d8\u96c6\u914d\u89c4\u5219', '\u710a\u63a5\u5de5\u827a\u89c4\u5219', '\u65e0\u635f\u68c0\u6d4b\u89c4\u5219', '\u7ba1\u7cfb\u8bd5\u538b\u89c4\u5219', '\u7535\u6c14\u8c03\u8bd5\u89c4\u5219', '\u8239\u7ea7\u793e\u68c0\u9a8c\u89c4\u5219', '\u4f9b\u5e94\u5546\u4ea4\u4ed8\u7ecf\u9a8c', '\u59ca\u59b9\u8239\u5386\u53f2\u5de5\u65f6', '\u8bbe\u5907\u6545\u969c\u7ecf\u9a8c', '\u5178\u578b\u5ef6\u671f\u6848\u4f8b', '\u6392\u4ea7\u7b56\u7565\u8bf4\u660e', '\u5f02\u5e38\u91cd\u6392\u5904\u7f6e\u89c4\u5219', ) _REQUIRED_DATA: dict[str, tuple[str, ...]] = { "ship-project-master-planning": ("contracts", "ship-projects", "milestones", "calendars"), "ship-wbs-network-planning": ("wbs", "blocks", "zones", "work-packages", "milestones"), "ship-bom-explosion": ("ebom", "pbom", "mbom", "materials", "work-packages"), "ship-mrp-netting": ("material-requirements", "inventory", "inventory-allocations", "planned-receipts", "substitutes"), "ship-make-buy-outsource-decision": ("materials", "routings", "suppliers", "material-requirements"), "ship-material-readiness": ("material-requirements", "kit-readiness", "inventory", "planned-receipts"), "ship-block-production-scheduling": ("operations", "resources", "teams", "calendars", "kit-readiness"), "ship-dock-erection-scheduling": ("blocks", "operations", "resources", "milestones", "calendars"), "ship-crane-lift-scheduling": ("blocks", "resources", "operations", "calendars"), "ship-yard-space-scheduling": ("blocks", "zones", "resources", "operations"), "ship-outfitting-zone-scheduling": ("zones", "work-packages", "operations", "teams"), "ship-workforce-scheduling": ("teams", "employees", "skills", "shifts", "operations"), "ship-procurement-planning": ("purchase-suggestions", "suppliers", "material-requirements", "planned-receipts"), "ship-outsourcing-planning": ("outsource-suggestions", "suppliers", "operations", "calendars"), "ship-quality-hold-planning": ("quality-inspections", "nonconformities", "rework-orders", "operations"), "ship-scenario-simulation": ("schedule-versions", "schedule-slots", "conflicts", "kpis"), "ship-schedule-repair": ("schedule-versions", "schedule-slots", "conflicts", "milestones"), "ship-schedule-explanation": ("schedule-versions", "schedule-slots", "kpis", "conflicts"), "ship-bottleneck-detection": ("resource-loads", "schedule-slots", "conflicts", "material-requirements"), "ship-risk-prediction": ("milestones", "conflicts", "kpis", "material-requirements", "nonconformities"), } _ALGORITHM_PROFILES: tuple[tuple[str, ...], ...] = ( ("CPM/PERT", "CRITICAL_CHAIN"), ("CPM/PERT", "RCPSP"), ("DETERMINISTIC_BOM_EXPLOSION",), ("DETERMINISTIC_MRP_NETTING", "ROLLING_HORIZON"), ("MILP", "BOTTLENECK_HEURISTIC"), ("DETERMINISTIC_KITTING", "ROLLING_HORIZON"), ("MULTI_MODE_RCPSP", "CP-SAT", "LARGE_NEIGHBORHOOD_SEARCH"), ("CP-SAT", "RCPSP", "CRITICAL_CHAIN"), ("CRANE_SEQUENCE_OPTIMIZATION", "CP-SAT", "TABU_SEARCH"), ("YARD_SPACE_OPTIMIZATION", "SIMULATED_ANNEALING", "GENETIC_ALGORITHM"), ("RCPSP", "LARGE_NEIGHBORHOOD_SEARCH"), ("CP-SAT", "MILP", "ROLLING_HORIZON"), ("DETERMINISTIC_PROCUREMENT_NETTING", "MILP"), ("ROLLING_HORIZON", "MILP", "LOCAL_REPAIR"), ("CONSTRAINT_PROPAGATION", "CP-SAT"), ("NSGA-II", "GENETIC_ALGORITHM", "SIMULATED_ANNEALING"), ("LOCAL_REPAIR", "LARGE_NEIGHBORHOOD_SEARCH", "ROLLING_HORIZON"), ("EVIDENCE_GRAPH", "RULE_EXPLAINER"), ("BOTTLENECK_HEURISTIC", "CRITICAL_CHAIN"), ("PERT_RISK_SIMULATION", "MONTE_CARLO", "RULE_SCORING"), ) def _skill_schema(skill_id: str, title: str) -> dict[str, Any]: evidence_refs = { "type": "array", "items": {"type": "string", "minLength": 1}, "minItems": 1, } return { "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": f"urn:aps:shipyard:skill:{skill_id}:1.0.0", "title": f"{title} input/output schema", "$defs": { "input": { "type": "object", "additionalProperties": True, "required": ["datasetVersion", "scenarioId", "inputDigest"], "properties": { "datasetVersion": {"type": "string", "minLength": 1}, "scenarioId": {"type": "string", "minLength": 1}, "inputDigest": {"type": "string", "minLength": 1}, "parameters": {"type": "object"}, "evidenceRefs": evidence_refs, }, }, "output": { "type": "object", "additionalProperties": True, "required": [ "algorithm", "inputSummary", "constraintCounts", "hardViolationCount", "softConstraintCost", "solveStatus", "solveTimeMs", "optimalityGap", "gapType", "fallbackReason", "evidenceRefs", ], "properties": { "algorithm": {"type": "string", "minLength": 1}, "inputSummary": {"type": "object"}, "constraintCounts": {"type": "object"}, "hardViolationCount": {"type": "integer", "minimum": 0}, "softConstraintCost": {"type": "object"}, "solveStatus": {"enum": ["FEASIBLE", "INFEASIBLE", "UNKNOWN"]}, "solveTimeMs": {"type": "integer", "minimum": 0}, "optimalityGap": {"type": ["number", "null"]}, "gapType": {"enum": ["ABSOLUTE", "RELATIVE", "NOT_APPLICABLE"]}, "fallbackReason": {"type": ["string", "null"]}, "evidenceRefs": evidence_refs, "result": {}, }, }, }, } def _sample_values(bundle: DatasetBundle, table: str, fields: tuple[str, ...]) -> list[str]: values: list[str] = [] for row in bundle.rows(table): for field in fields: value = row.get(field) if value not in (None, ""): values.append(str(value)) break return sorted(set(values)) def _build_skills() -> tuple[list[dict[str, Any]], dict[str, dict[str, Any]]]: skills: list[dict[str, Any]] = [] schemas: dict[str, dict[str, Any]] = {} for index, skill_id in enumerate(SKILL_IDS): algorithms = _ALGORITHM_PROFILES[index] schema_path = f"skills/skill-schemas/{skill_id}.schema.json" title = skill_id.removeprefix("ship-").replace("-", " ").title() skill = { "skillId": skill_id, "name": title, "description": f"Deterministic synthetic APS skill for {title.lower()}.", "inputSchema": f"{schema_path}#/$defs/input", "outputSchema": f"{schema_path}#/$defs/output", "requiredData": list(_REQUIRED_DATA[skill_id]), "algorithmCandidates": list(algorithms), "hardConstraints": [ "VERSIONED_INPUT", "STRUCTURED_HARD_CONSTRAINTS", "RESOLVABLE_EVIDENCE_REFS", ], "softConstraints": ["MINIMIZE_LATENESS", "MINIMIZE_PLAN_DISRUPTION"], "fallbackAlgorithm": algorithms[-1], "timeoutSeconds": 600 if "scheduling" in skill_id or "simulation" in skill_id else 180, "validationRules": [ "INPUT_DIGEST_REQUIRED", "HARD_VIOLATION_COUNT_REQUIRED", "EVIDENCE_REFS_MUST_RESOLVE", "HEURISTIC_GAP_MUST_BE_NOT_APPLICABLE", ], "evidenceFields": [ "algorithm", "inputDigest", "constraintCounts", "solveStatus", "solveTimeMs", "optimalityGap", "gapType", "fallbackReason", "evidenceRefs", ], "version": GENERATOR_VERSION, "evidenceRef": f"skill:{skill_id}", } skills.append(skill) schemas[skill_id] = _skill_schema(skill_id, title) return skills, schemas def _build_knowledge_assets(bundle: DatasetBundle, config: GeneratorConfig) -> list[dict[str, Any]]: per_category = {"small": 1, "standard": 3, "full": 6}[config.scale] ship_types = _sample_values(bundle, "ship-projects", ("shipType",)) or [ "SYNTHETIC_GENERAL_SHIP" ] workshop_ids = _sample_values(bundle, "workshops", ("workshopId", "name")) or [ "SYNTHETIC_WORKSHOP" ] resource_ids = _sample_values(bundle, "resources", ("resourceId",)) material_groups = _sample_values(bundle, "materials", ("materialGroup", "category", "materialType")) operation_codes = _sample_values(bundle, "operations", ("operationCode", "code")) or _sample_values( bundle, "routing-operations", ("operationCode", "code"), ) assets: list[dict[str, Any]] = [] for category_index, category in enumerate(RAG_CATEGORIES, start=1): skill_id = SKILL_IDS[category_index - 1] for item_index in range(1, per_category + 1): knowledge_id = f"KNO-SYN-{category_index:02d}-{item_index:03d}" selector = category_index + item_index - 2 assets.append( { "knowledgeId": knowledge_id, "title": f"Synthetic rule {category_index:02d}-{item_index:03d}", "category": category, "content": ( "Synthetic knowledge for APS development and validation. " "Execution must rely on structured constraints, resource calendars, and versioned inputs. " "This is not a real Beihai Shipyard policy, experience record, or historical fact." ), "applicableShipTypes": ship_types, "applicableWorkshops": [workshop_ids[selector % len(workshop_ids)]], "tags": ["SYNTHETIC", "APS", f"CATEGORY_{category_index:02d}", skill_id], "sourceType": "SYNTHETIC_KNOWLEDGE", "version": GENERATOR_VERSION, "effectiveDate": config.planning_base_date.isoformat(), "confidence": round(0.72 + (item_index / 100), 2), "relatedResourceIds": [resource_ids[selector % len(resource_ids)]] if resource_ids else [], "relatedMaterialGroups": [material_groups[selector % len(material_groups)]] if material_groups else [], "relatedOperationCodes": [operation_codes[selector % len(operation_codes)]] if operation_codes else [], "evidenceRef": knowledge_id, "relatedSkillId": skill_id, "relatedSkillEvidenceRef": f"skill:{skill_id}", } ) return assets def _rag_markdown(assets: list[dict[str, Any]]) -> dict[str, str]: by_category: dict[str, list[dict[str, Any]]] = {category: [] for category in RAG_CATEGORIES} for asset in assets: by_category[str(asset["category"])].append(asset) def render(title: str, categories: tuple[str, ...]) -> str: lines = [f"# {title}", "", f"> {DATA_DISCLAIMER}", ""] for category in categories: lines.extend((f"## {category}", "")) for asset in by_category[category]: lines.extend( ( f"### {asset['title']}", "", str(asset["content"]), "", f"- evidenceRef: `{asset['evidenceRef']}`", f"- skillEvidenceRef: `{asset['relatedSkillEvidenceRef']}`", f"- confidence: `{asset['confidence']}`", "", ) ) return "\n".join(lines).rstrip() + "\n" return { "shipbuilding-rules.md": render("Synthetic Shipbuilding Rules", RAG_CATEGORIES[:10]), "scheduling-rules.md": render("Synthetic Scheduling Rules", RAG_CATEGORIES[2:10] + RAG_CATEGORIES[18:20]), "quality-rules.md": render("Synthetic Quality Rules", RAG_CATEGORIES[9:14]), "historical-lessons.md": render("Synthetic Historical Lessons", RAG_CATEGORIES[14:20]), } def generate_rag_and_skills(bundle: DatasetBundle, config: GeneratorConfig) -> DatasetBundle: """Populate deterministic synthetic RAG assets and the exact 20 APS skills.""" skills, skill_schemas = _build_skills() knowledge_assets = _build_knowledge_assets(bundle, config) bundle.artifacts["skill-registry"] = { "datasetType": "SYNTHETIC", "generatedFor": "APS development and validation", "version": GENERATOR_VERSION, "skillCount": len(skills), "skills": skills, } bundle.artifacts["skill-schemas"] = skill_schemas bundle.artifacts["knowledge-assets"] = knowledge_assets bundle.artifacts["rag-markdown"] = _rag_markdown(knowledge_assets) return bundle