close guardrail gaps and fix the scoring pipeline top to bottom
Constraint resolver: aggregate mass/footprint across a combo instead of pairwise-only checks, treat medium/atmosphere as agreement not supply/demand, reduce multi-provider checks by best/sum instead of AND-ing every provider, fail closed on unrecognized mutex values, add a propulsion-viability (thrust-to-weight) rule. Seed data updated to match (nuclear/solar-sail footprint floors, water-medium exclusions, explicit ground/gravity providers). Domain metric units were stored globally per metric name instead of per-domain, silently corrupting cost_efficiency for every domain but the first one seeded — fixed with a schema migration. Stub estimator's cost_efficiency/safety/availability/reliability were a backwards formula and flat constants; replaced with heuristics grounded in each entity's thrust_profile/energy_form/infrastructure. LLM estimate_physics() now receives each metric's unit and expected range instead of a bare name, fixing wildly miscalibrated estimates traced back to the prompt's own hardcoded example anchoring the model to the wrong order of magnitude. Sharpened the safety-estimation and plausibility-review prompts. Deduped provider parsing logic into llm/parsing.py. Web pipeline form can now pick an LLM provider per run instead of only via server env var. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -233,6 +233,16 @@ class Repository:
|
|||||||
)
|
)
|
||||||
self.conn.commit()
|
self.conn.commit()
|
||||||
|
|
||||||
|
def backfill_metric_unit(self, domain_name: str, metric_name: str, unit: str) -> None:
|
||||||
|
"""Set this domain-metric row's unit — unit is domain-scoped, not global to the metric name."""
|
||||||
|
self.conn.execute(
|
||||||
|
"""UPDATE domain_metric_weights SET unit = ?
|
||||||
|
WHERE domain_id = (SELECT id FROM domains WHERE name = ?)
|
||||||
|
AND metric_id = (SELECT id FROM metrics WHERE name = ?)""",
|
||||||
|
(unit, domain_name, metric_name),
|
||||||
|
)
|
||||||
|
self.conn.commit()
|
||||||
|
|
||||||
def add_domain(self, domain: Domain) -> Domain:
|
def add_domain(self, domain: Domain) -> Domain:
|
||||||
cur = self.conn.execute(
|
cur = self.conn.execute(
|
||||||
"INSERT INTO domains (name, description) VALUES (?, ?)",
|
"INSERT INTO domains (name, description) VALUES (?, ?)",
|
||||||
@@ -244,10 +254,10 @@ class Repository:
|
|||||||
mb.metric_id = metric_id
|
mb.metric_id = metric_id
|
||||||
self.conn.execute(
|
self.conn.execute(
|
||||||
"""INSERT INTO domain_metric_weights
|
"""INSERT INTO domain_metric_weights
|
||||||
(domain_id, metric_id, weight, norm_min, norm_max, lower_is_better)
|
(domain_id, metric_id, weight, norm_min, norm_max, lower_is_better, unit)
|
||||||
VALUES (?, ?, ?, ?, ?, ?)""",
|
VALUES (?, ?, ?, ?, ?, ?, ?)""",
|
||||||
(domain.id, metric_id, mb.weight, mb.norm_min, mb.norm_max,
|
(domain.id, metric_id, mb.weight, mb.norm_min, mb.norm_max,
|
||||||
int(mb.lower_is_better)),
|
int(mb.lower_is_better), mb.unit),
|
||||||
)
|
)
|
||||||
for dc in domain.constraints:
|
for dc in domain.constraints:
|
||||||
for val in dc.allowed_values:
|
for val in dc.allowed_values:
|
||||||
@@ -273,7 +283,7 @@ class Repository:
|
|||||||
if not row:
|
if not row:
|
||||||
return None
|
return None
|
||||||
weights = self.conn.execute(
|
weights = self.conn.execute(
|
||||||
"""SELECT m.name, m.unit, dmw.weight, dmw.norm_min, dmw.norm_max,
|
"""SELECT m.name, dmw.unit, dmw.weight, dmw.norm_min, dmw.norm_max,
|
||||||
dmw.metric_id, dmw.lower_is_better
|
dmw.metric_id, dmw.lower_is_better
|
||||||
FROM domain_metric_weights dmw
|
FROM domain_metric_weights dmw
|
||||||
JOIN metrics m ON dmw.metric_id = m.id
|
JOIN metrics m ON dmw.metric_id = m.id
|
||||||
@@ -318,10 +328,10 @@ class Repository:
|
|||||||
mb.metric_id = metric_id
|
mb.metric_id = metric_id
|
||||||
self.conn.execute(
|
self.conn.execute(
|
||||||
"""INSERT OR REPLACE INTO domain_metric_weights
|
"""INSERT OR REPLACE INTO domain_metric_weights
|
||||||
(domain_id, metric_id, weight, norm_min, norm_max, lower_is_better)
|
(domain_id, metric_id, weight, norm_min, norm_max, lower_is_better, unit)
|
||||||
VALUES (?, ?, ?, ?, ?, ?)""",
|
VALUES (?, ?, ?, ?, ?, ?, ?)""",
|
||||||
(domain_id, metric_id, mb.weight, mb.norm_min, mb.norm_max,
|
(domain_id, metric_id, mb.weight, mb.norm_min, mb.norm_max,
|
||||||
int(mb.lower_is_better)),
|
int(mb.lower_is_better), mb.unit),
|
||||||
)
|
)
|
||||||
self.conn.commit()
|
self.conn.commit()
|
||||||
return mb
|
return mb
|
||||||
@@ -332,14 +342,9 @@ class Repository:
|
|||||||
) -> None:
|
) -> None:
|
||||||
self.conn.execute(
|
self.conn.execute(
|
||||||
"""UPDATE domain_metric_weights
|
"""UPDATE domain_metric_weights
|
||||||
SET weight = ?, norm_min = ?, norm_max = ?, lower_is_better = ?
|
SET weight = ?, norm_min = ?, norm_max = ?, lower_is_better = ?, unit = ?
|
||||||
WHERE domain_id = ? AND metric_id = ?""",
|
WHERE domain_id = ? AND metric_id = ?""",
|
||||||
(weight, norm_min, norm_max, int(lower_is_better), domain_id, metric_id),
|
(weight, norm_min, norm_max, int(lower_is_better), unit, domain_id, metric_id),
|
||||||
)
|
|
||||||
if unit:
|
|
||||||
self.conn.execute(
|
|
||||||
"UPDATE metrics SET unit = ? WHERE id = ?",
|
|
||||||
(unit, metric_id),
|
|
||||||
)
|
)
|
||||||
self.conn.commit()
|
self.conn.commit()
|
||||||
|
|
||||||
|
|||||||
@@ -51,6 +51,7 @@ CREATE TABLE IF NOT EXISTS domain_metric_weights (
|
|||||||
norm_min REAL,
|
norm_min REAL,
|
||||||
norm_max REAL,
|
norm_max REAL,
|
||||||
lower_is_better INTEGER NOT NULL DEFAULT 0,
|
lower_is_better INTEGER NOT NULL DEFAULT 0,
|
||||||
|
unit TEXT,
|
||||||
UNIQUE(domain_id, metric_id)
|
UNIQUE(domain_id, metric_id)
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -134,6 +135,16 @@ def _migrate(conn: sqlite3.Connection) -> None:
|
|||||||
conn.execute(
|
conn.execute(
|
||||||
"ALTER TABLE domain_metric_weights ADD COLUMN lower_is_better INTEGER NOT NULL DEFAULT 0"
|
"ALTER TABLE domain_metric_weights ADD COLUMN lower_is_better INTEGER NOT NULL DEFAULT 0"
|
||||||
)
|
)
|
||||||
|
if "unit" not in cols:
|
||||||
|
conn.execute("ALTER TABLE domain_metric_weights ADD COLUMN unit TEXT")
|
||||||
|
# Best-effort backfill from the old (metric-name-global) unit column —
|
||||||
|
# only correct for domains that happen to agree on that metric's unit.
|
||||||
|
# Seed data re-applies each domain's real per-domain unit on next load.
|
||||||
|
conn.execute(
|
||||||
|
"""UPDATE domain_metric_weights
|
||||||
|
SET unit = (SELECT m.unit FROM metrics m WHERE m.id = domain_metric_weights.metric_id)
|
||||||
|
WHERE unit IS NULL"""
|
||||||
|
)
|
||||||
|
|
||||||
# Create domain_constraints table if missing (added after initial schema)
|
# Create domain_constraints table if missing (added after initial schema)
|
||||||
tables = {r[0] for r in conn.execute(
|
tables = {r[0] for r in conn.execute(
|
||||||
|
|||||||
@@ -16,23 +16,42 @@ MUTEX_VALUES: dict[str, list[set[str]]] = {
|
|||||||
"medium": [{"ground"}, {"water"}, {"air"}, {"space"}],
|
"medium": [{"ground"}, {"water"}, {"air"}, {"space"}],
|
||||||
}
|
}
|
||||||
|
|
||||||
# Conditions assumed always available (don't need an explicit provides)
|
# Conditions assumed always available (don't need an explicit provides).
|
||||||
|
# ground_surface and gravity are deliberately NOT here — unlike star_proximity
|
||||||
|
# (only relevant to space-adjacent entities) they're things most, but not all,
|
||||||
|
# platforms actually have (a Spaceship in orbital freefall has neither in the
|
||||||
|
# sense a ground-rolling actuator needs); those platforms must `provide` them.
|
||||||
AMBIENT_CONDITIONS: set[tuple[str, str]] = {
|
AMBIENT_CONDITIONS: set[tuple[str, str]] = {
|
||||||
("ground_surface", "true"),
|
|
||||||
("gravity", "true"),
|
|
||||||
("star_proximity", "true"),
|
("star_proximity", "true"),
|
||||||
|
("water_surface", "true"),
|
||||||
}
|
}
|
||||||
|
|
||||||
# Per-category behavior for unmet requirements:
|
# Per-category behavior for unmet requirements:
|
||||||
# "block" = hard violation, "warn" = conditional warning, "skip" = ignore
|
# "block" = hard violation, "warn" = conditional warning, "skip" = ignore
|
||||||
CATEGORY_SEVERITY: dict[str, str] = {
|
CATEGORY_SEVERITY: dict[str, str] = {
|
||||||
"energy": "block",
|
"energy": "block",
|
||||||
|
"environment": "block",
|
||||||
"infrastructure": "skip",
|
"infrastructure": "skip",
|
||||||
}
|
}
|
||||||
|
|
||||||
# For provides-vs-range_min: deficit > this ratio = hard block, else warning
|
# For provides-vs-range_min: deficit > this ratio = hard block, else warning
|
||||||
DEFICIT_THRESHOLD: float = 0.25
|
DEFICIT_THRESHOLD: float = 0.25
|
||||||
|
|
||||||
|
# How multiple entities' numbers on the same key combine into one system-level
|
||||||
|
# number. "sum" = extensive (component contributions add into one vehicle);
|
||||||
|
# any key not listed defaults to "max" (today's pairwise behavior — the
|
||||||
|
# strongest/most-demanding single entity wins).
|
||||||
|
KEY_AGGREGATION: dict[str, str] = {
|
||||||
|
"mass": "sum",
|
||||||
|
"footprint": "sum",
|
||||||
|
}
|
||||||
|
|
||||||
|
# Sum-of-floors is an estimate built from independent component minima, not a
|
||||||
|
# measurement. Overrun inside this band warns; beyond it blocks.
|
||||||
|
# ponytail: single global tolerance; per-key band if mass and footprint ever
|
||||||
|
# need different slack.
|
||||||
|
OVERRUN_TOLERANCE: float = 0.10
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class ConstraintResult:
|
class ConstraintResult:
|
||||||
@@ -52,6 +71,8 @@ class ConstraintResolver:
|
|||||||
ambient_conditions=None,
|
ambient_conditions=None,
|
||||||
category_severity=None,
|
category_severity=None,
|
||||||
deficit_threshold=None,
|
deficit_threshold=None,
|
||||||
|
key_aggregation=None,
|
||||||
|
overrun_tolerance=None,
|
||||||
) -> None:
|
) -> None:
|
||||||
self.mutex = mutex_registry or MUTEX_VALUES
|
self.mutex = mutex_registry or MUTEX_VALUES
|
||||||
self.ambient = ambient_conditions or AMBIENT_CONDITIONS
|
self.ambient = ambient_conditions or AMBIENT_CONDITIONS
|
||||||
@@ -59,6 +80,10 @@ class ConstraintResolver:
|
|||||||
self.deficit_threshold = (
|
self.deficit_threshold = (
|
||||||
deficit_threshold if deficit_threshold is not None else DEFICIT_THRESHOLD
|
deficit_threshold if deficit_threshold is not None else DEFICIT_THRESHOLD
|
||||||
)
|
)
|
||||||
|
self.key_aggregation = key_aggregation or KEY_AGGREGATION
|
||||||
|
self.overrun_tolerance = (
|
||||||
|
overrun_tolerance if overrun_tolerance is not None else OVERRUN_TOLERANCE
|
||||||
|
)
|
||||||
|
|
||||||
def resolve(self, combination: Combination) -> ConstraintResult:
|
def resolve(self, combination: Combination) -> ConstraintResult:
|
||||||
result = ConstraintResult()
|
result = ConstraintResult()
|
||||||
@@ -72,6 +97,7 @@ class ConstraintResolver:
|
|||||||
self._check_range_incompatibility(all_deps, result)
|
self._check_range_incompatibility(all_deps, result)
|
||||||
self._check_provides_vs_range(combination, result)
|
self._check_provides_vs_range(combination, result)
|
||||||
self._check_unmet_requirements(all_deps, result)
|
self._check_unmet_requirements(all_deps, result)
|
||||||
|
self._check_propulsion_viability(combination, result)
|
||||||
|
|
||||||
if result.violations:
|
if result.violations:
|
||||||
result.status = "p1_fail"
|
result.status = "p1_fail"
|
||||||
@@ -91,11 +117,24 @@ class ConstraintResolver:
|
|||||||
for exc_name, exc in excludes:
|
for exc_name, exc in excludes:
|
||||||
if req_name == exc_name:
|
if req_name == exc_name:
|
||||||
continue
|
continue
|
||||||
if req.key == exc.key and req.value == exc.value:
|
if req.key != exc.key:
|
||||||
|
continue
|
||||||
|
if req.value == exc.value:
|
||||||
result.violations.append(
|
result.violations.append(
|
||||||
f"{req_name} requires {req.key}={req.value} "
|
f"{req_name} requires {req.key}={req.value} "
|
||||||
f"but {exc_name} excludes it"
|
f"but {exc_name} excludes it"
|
||||||
)
|
)
|
||||||
|
elif req.key in self.mutex:
|
||||||
|
# Excluding one value in a mutex family excludes the
|
||||||
|
# whole family (e.g. excludes atmosphere=standard also
|
||||||
|
# rules out other "dense" values in the same set).
|
||||||
|
exc_set = self._find_mutex_set(exc.key, exc.value)
|
||||||
|
if exc_set is not None and req.value in exc_set:
|
||||||
|
result.violations.append(
|
||||||
|
f"{req_name} requires {req.key}={req.value} "
|
||||||
|
f"but {exc_name} excludes {exc.key}={exc.value} "
|
||||||
|
f"(same mutex family)"
|
||||||
|
)
|
||||||
|
|
||||||
def _check_mutual_exclusion(
|
def _check_mutual_exclusion(
|
||||||
self, all_deps: list[tuple[str, Dependency]], result: ConstraintResult
|
self, all_deps: list[tuple[str, Dependency]], result: ConstraintResult
|
||||||
@@ -111,11 +150,14 @@ class ConstraintResolver:
|
|||||||
continue
|
continue
|
||||||
if dep_a.value == dep_b.value:
|
if dep_a.value == dep_b.value:
|
||||||
continue
|
continue
|
||||||
# Check if values are in different mutex sets
|
# Check if values are in different mutex sets. An unrecognized
|
||||||
|
# value (not in any registered set) is treated as conflicting
|
||||||
|
# with any recognized value on the same key — fail closed
|
||||||
|
# rather than silently letting unknown values through.
|
||||||
if dep_a.key in self.mutex:
|
if dep_a.key in self.mutex:
|
||||||
set_a = self._find_mutex_set(dep_a.key, dep_a.value)
|
set_a = self._find_mutex_set(dep_a.key, dep_a.value)
|
||||||
set_b = self._find_mutex_set(dep_b.key, dep_b.value)
|
set_b = self._find_mutex_set(dep_b.key, dep_b.value)
|
||||||
if set_a is not None and set_b is not None and set_a is not set_b:
|
if set_a is not set_b:
|
||||||
result.violations.append(
|
result.violations.append(
|
||||||
f"{name_a} requires {dep_a.key}={dep_a.value} "
|
f"{name_a} requires {dep_a.key}={dep_a.value} "
|
||||||
f"but {name_b} requires {dep_b.key}={dep_b.value} "
|
f"but {name_b} requires {dep_b.key}={dep_b.value} "
|
||||||
@@ -132,7 +174,13 @@ class ConstraintResolver:
|
|||||||
def _check_range_incompatibility(
|
def _check_range_incompatibility(
|
||||||
self, all_deps: list[tuple[str, Dependency]], result: ConstraintResult
|
self, all_deps: list[tuple[str, Dependency]], result: ConstraintResult
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Rule 3: If A range_min > B range_max for the same key → BLOCKED."""
|
"""Rule 3: floors on a key must fit under the tightest ceiling on that key.
|
||||||
|
|
||||||
|
Extensive keys ("sum" in key_aggregation) aggregate every entity's
|
||||||
|
floor before the comparison, since they represent components sharing
|
||||||
|
one physical vehicle (mass, footprint). Every other key keeps the
|
||||||
|
original pairwise floor-vs-ceiling check.
|
||||||
|
"""
|
||||||
range_mins: dict[str, list[tuple[str, float]]] = {}
|
range_mins: dict[str, list[tuple[str, float]]] = {}
|
||||||
range_maxs: dict[str, list[tuple[str, float]]] = {}
|
range_maxs: dict[str, list[tuple[str, float]]] = {}
|
||||||
|
|
||||||
@@ -143,6 +191,7 @@ class ConstraintResolver:
|
|||||||
range_maxs.setdefault(dep.key, []).append((name, float(dep.value)))
|
range_maxs.setdefault(dep.key, []).append((name, float(dep.value)))
|
||||||
|
|
||||||
for key in set(range_mins) & set(range_maxs):
|
for key in set(range_mins) & set(range_maxs):
|
||||||
|
if self.key_aggregation.get(key) != "sum":
|
||||||
for min_name, min_val in range_mins[key]:
|
for min_name, min_val in range_mins[key]:
|
||||||
for max_name, max_val in range_maxs[key]:
|
for max_name, max_val in range_maxs[key]:
|
||||||
if min_name == max_name:
|
if min_name == max_name:
|
||||||
@@ -152,11 +201,33 @@ class ConstraintResolver:
|
|||||||
f"{min_name} requires {key} >= {min_val} "
|
f"{min_name} requires {key} >= {min_val} "
|
||||||
f"but {max_name} limits {key} <= {max_val}"
|
f"but {max_name} limits {key} <= {max_val}"
|
||||||
)
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
total = sum(val for _, val in range_mins[key])
|
||||||
|
ceil_name, ceiling = min(range_maxs[key], key=lambda t: t[1])
|
||||||
|
if total <= ceiling:
|
||||||
|
continue
|
||||||
|
parts = ", ".join(f"{name} {val:g}" for name, val in range_mins[key])
|
||||||
|
msg = (
|
||||||
|
f"combined {key} {total:g} ({parts}) exceeds "
|
||||||
|
f"{ceil_name} limit of {ceiling:g}"
|
||||||
|
)
|
||||||
|
if total > ceiling * (1 + self.overrun_tolerance):
|
||||||
|
result.violations.append(msg)
|
||||||
|
else:
|
||||||
|
result.warnings.append(msg)
|
||||||
|
|
||||||
def _check_provides_vs_range(
|
def _check_provides_vs_range(
|
||||||
self, combination: Combination, result: ConstraintResult
|
self, combination: Combination, result: ConstraintResult
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Generic: provides(key, N) < range_min(key, M) → block/warn."""
|
"""Generic: provides(key, N) < range_min(key, M) → block/warn.
|
||||||
|
|
||||||
|
Multiple providers of the same key are reduced to one number before
|
||||||
|
comparing: summed for extensive keys, otherwise the strongest single
|
||||||
|
provider wins (a weak secondary source, e.g. backup solar panels
|
||||||
|
alongside a nuclear reactor, must not drag down a combo that's
|
||||||
|
already satisfied by its best provider).
|
||||||
|
"""
|
||||||
provided: dict[str, list[tuple[str, float]]] = {}
|
provided: dict[str, list[tuple[str, float]]] = {}
|
||||||
required: dict[str, list[tuple[str, float]]] = {}
|
required: dict[str, list[tuple[str, float]]] = {}
|
||||||
|
|
||||||
@@ -172,8 +243,13 @@ class ConstraintResolver:
|
|||||||
required.setdefault(dep.key, []).append((entity.name, val))
|
required.setdefault(dep.key, []).append((entity.name, val))
|
||||||
|
|
||||||
for key in set(provided) & set(required):
|
for key in set(provided) & set(required):
|
||||||
|
if self.key_aggregation.get(key) == "sum":
|
||||||
|
prov_name = " + ".join(name for name, _ in provided[key])
|
||||||
|
prov_val = sum(val for _, val in provided[key])
|
||||||
|
else:
|
||||||
|
prov_name, prov_val = max(provided[key], key=lambda t: t[1])
|
||||||
|
|
||||||
for req_name, req_val in required[key]:
|
for req_name, req_val in required[key]:
|
||||||
for prov_name, prov_val in provided[key]:
|
|
||||||
if prov_val < req_val * self.deficit_threshold:
|
if prov_val < req_val * self.deficit_threshold:
|
||||||
result.violations.append(
|
result.violations.append(
|
||||||
f"{prov_name} provides {key}={prov_val:.0f} but "
|
f"{prov_name} provides {key}={prov_val:.0f} but "
|
||||||
@@ -206,6 +282,86 @@ class ConstraintResolver:
|
|||||||
result.status = "p1_fail"
|
result.status = "p1_fail"
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
def _check_propulsion_viability(
|
||||||
|
self, combination: Combination, result: ConstraintResult
|
||||||
|
) -> None:
|
||||||
|
"""Rule 6: an entity providing specific_thrust (N/kg of its own mass)
|
||||||
|
must be able to mass enough, within the vehicle's mass budget, to
|
||||||
|
accelerate the whole combo past the platform's min_effective_accel
|
||||||
|
(m/s²) — the same physics whether that's overcoming rolling
|
||||||
|
resistance or hovering against gravity, just different constants.
|
||||||
|
|
||||||
|
Skips silently if no entity declares min_effective_accel or no
|
||||||
|
entity declares specific_thrust — this only fires where both
|
||||||
|
numbers are actually known.
|
||||||
|
"""
|
||||||
|
min_accel = next(
|
||||||
|
(
|
||||||
|
float(dep.value)
|
||||||
|
for entity in combination.entities
|
||||||
|
for dep in entity.dependencies
|
||||||
|
if dep.key == "min_effective_accel" and dep.constraint_type == "range_min"
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if min_accel is None:
|
||||||
|
return
|
||||||
|
|
||||||
|
for actuator in combination.entities:
|
||||||
|
specific_thrust = next(
|
||||||
|
(
|
||||||
|
float(dep.value)
|
||||||
|
for dep in actuator.dependencies
|
||||||
|
if dep.key == "specific_thrust" and dep.constraint_type == "provides"
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if specific_thrust is None:
|
||||||
|
continue
|
||||||
|
|
||||||
|
other_mass = sum(
|
||||||
|
float(dep.value)
|
||||||
|
for entity in combination.entities
|
||||||
|
if entity is not actuator
|
||||||
|
for dep in entity.dependencies
|
||||||
|
if dep.key == "mass" and dep.constraint_type == "range_min"
|
||||||
|
)
|
||||||
|
|
||||||
|
if specific_thrust <= min_accel:
|
||||||
|
result.violations.append(
|
||||||
|
f"{actuator.name} specific thrust {specific_thrust:g} N/kg can "
|
||||||
|
f"never exceed the {min_accel:g} m/s² minimum this vehicle "
|
||||||
|
f"needs, regardless of scale"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
required_mass = min_accel * other_mass / (specific_thrust - min_accel)
|
||||||
|
actuator_floor = next(
|
||||||
|
(
|
||||||
|
float(dep.value)
|
||||||
|
for dep in actuator.dependencies
|
||||||
|
if dep.key == "mass" and dep.constraint_type == "range_min"
|
||||||
|
),
|
||||||
|
0.0,
|
||||||
|
)
|
||||||
|
effective_mass = max(required_mass, actuator_floor)
|
||||||
|
|
||||||
|
ceiling = next(
|
||||||
|
(
|
||||||
|
float(dep.value)
|
||||||
|
for entity in combination.entities
|
||||||
|
for dep in entity.dependencies
|
||||||
|
if dep.key == "mass" and dep.constraint_type == "range_max"
|
||||||
|
),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
if ceiling is not None and effective_mass + other_mass > ceiling:
|
||||||
|
result.violations.append(
|
||||||
|
f"{actuator.name} would need >= {effective_mass:.0f}kg to move "
|
||||||
|
f"this vehicle at {min_accel:g} m/s², exceeding its "
|
||||||
|
f"{ceiling:g}kg mass ceiling"
|
||||||
|
)
|
||||||
|
|
||||||
def _check_unmet_requirements(
|
def _check_unmet_requirements(
|
||||||
self, all_deps: list[tuple[str, Dependency]], result: ConstraintResult
|
self, all_deps: list[tuple[str, Dependency]], result: ConstraintResult
|
||||||
) -> None:
|
) -> None:
|
||||||
@@ -215,6 +371,11 @@ class ConstraintResolver:
|
|||||||
for name, dep in all_deps:
|
for name, dep in all_deps:
|
||||||
if dep.constraint_type != "requires":
|
if dep.constraint_type != "requires":
|
||||||
continue
|
continue
|
||||||
|
if dep.key in self.mutex:
|
||||||
|
# Agreement key (e.g. medium, atmosphere) — entities must
|
||||||
|
# concur, not supply/demand. Rule 2 owns compatibility here;
|
||||||
|
# no entity is expected to "provide" it.
|
||||||
|
continue
|
||||||
severity = self.category_severity.get(dep.category, "warn")
|
severity = self.category_severity.get(dep.category, "warn")
|
||||||
if severity == "skip":
|
if severity == "skip":
|
||||||
continue
|
continue
|
||||||
|
|||||||
@@ -12,7 +12,113 @@ from physcom.engine.constraint_resolver import ConstraintResolver, ConstraintRes
|
|||||||
from physcom.engine.scorer import Scorer
|
from physcom.engine.scorer import Scorer
|
||||||
from physcom.llm.base import LLMProvider, LLMRateLimitError
|
from physcom.llm.base import LLMProvider, LLMRateLimitError
|
||||||
from physcom.models.combination import Combination, ScoredResult
|
from physcom.models.combination import Combination, ScoredResult
|
||||||
from physcom.models.domain import Domain
|
from physcom.models.domain import Domain, MetricBound
|
||||||
|
|
||||||
|
# Stub-estimator heuristics (used only when no LLM provider is configured).
|
||||||
|
# Keyed by the same categorical vocabulary already used in seed data — never
|
||||||
|
# by entity name, so new entities inherit sensible behavior automatically.
|
||||||
|
|
||||||
|
# How controllable a thrust delivery profile is — bursty/extreme profiles are
|
||||||
|
# harder to control and cost more per use (ammunition, propellant, wear) than
|
||||||
|
# steady ones. Missing values fall back to a neutral 1.0/0.6.
|
||||||
|
THRUST_PROFILE_COST_MULTIPLIER: dict[str, float] = {
|
||||||
|
"low_continuous": 1.0,
|
||||||
|
"continuous_low": 1.0,
|
||||||
|
"moderate_continuous": 1.1,
|
||||||
|
"high_continuous": 1.3,
|
||||||
|
"extreme_continuous": 1.6,
|
||||||
|
"high_burst": 2.5,
|
||||||
|
"extreme_burst": 4.0,
|
||||||
|
}
|
||||||
|
THRUST_PROFILE_SAFETY: dict[str, float] = {
|
||||||
|
"low_continuous": 0.9,
|
||||||
|
"continuous_low": 0.9,
|
||||||
|
"moderate_continuous": 0.75,
|
||||||
|
"high_continuous": 0.6,
|
||||||
|
"extreme_continuous": 0.45,
|
||||||
|
"high_burst": 0.3,
|
||||||
|
"extreme_burst": 0.15,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Baseline hazard of the energy form itself, independent of delivery profile.
|
||||||
|
ENERGY_FORM_SAFETY: dict[str, float] = {
|
||||||
|
"biological": 0.9,
|
||||||
|
"electrical": 0.85,
|
||||||
|
"wind": 0.9,
|
||||||
|
"gravitational": 0.9,
|
||||||
|
"radiation_pressure": 0.85,
|
||||||
|
"kinetic_stored": 0.7,
|
||||||
|
"pneumatic": 0.75,
|
||||||
|
"chemical_combustible": 0.6,
|
||||||
|
"ion_propellant": 0.75,
|
||||||
|
"chemical_propellant": 0.4,
|
||||||
|
"chemical_explosive": 0.3,
|
||||||
|
"nuclear_thermal": 0.35,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Rough $/m base cost by energy form — renewables/muscle power are ~free,
|
||||||
|
# consumables (propellant, ammunition, nuclear fuel) cost real money per use.
|
||||||
|
# This is a categorical placeholder, not a physics formula — energy_density
|
||||||
|
# (J/kg) can't give a $/m figure on its own since it says nothing about price.
|
||||||
|
ENERGY_FORM_BASE_COST: dict[str, float] = {
|
||||||
|
"wind": 1e-6,
|
||||||
|
"gravitational": 1e-6,
|
||||||
|
"radiation_pressure": 1e-6,
|
||||||
|
"electrical": 3e-5,
|
||||||
|
"kinetic_stored": 2e-5,
|
||||||
|
"biological": 5e-5,
|
||||||
|
"pneumatic": 4e-5,
|
||||||
|
"chemical_combustible": 8e-5,
|
||||||
|
"nuclear_thermal": 1e-3,
|
||||||
|
"ion_propellant": 2e-3,
|
||||||
|
"chemical_propellant": 5e-3,
|
||||||
|
"chemical_explosive": 8e-3,
|
||||||
|
}
|
||||||
|
|
||||||
|
# How available the required infrastructure/fuel supply chain is today.
|
||||||
|
# Multiple matches in one combo (e.g. a platform's road_network requirement
|
||||||
|
# plus a storage's fuel_infrastructure requirement) are averaged.
|
||||||
|
INFRASTRUCTURE_AVAILABILITY: dict[tuple[str, str], float] = {
|
||||||
|
("road_network", "true"): 0.95,
|
||||||
|
("rail_network", "true"): 0.8,
|
||||||
|
("runway", "true"): 0.5,
|
||||||
|
("tow_or_winch", "true"): 0.5,
|
||||||
|
("hyperloop_tube", "true"): 0.1,
|
||||||
|
("launch_facility", "true"): 0.05,
|
||||||
|
("fuel_infrastructure", "none"): 1.0,
|
||||||
|
("fuel_infrastructure", "fuel_station"): 0.95,
|
||||||
|
("fuel_infrastructure", "charging_station"): 0.85,
|
||||||
|
("fuel_infrastructure", "cng_station"): 0.5,
|
||||||
|
("fuel_infrastructure", "coal_supply"): 0.5,
|
||||||
|
("fuel_infrastructure", "hydrogen_station"): 0.25,
|
||||||
|
("fuel_infrastructure", "compressed_air_station"): 0.3,
|
||||||
|
("fuel_infrastructure", "ammunition"): 0.3,
|
||||||
|
("fuel_infrastructure", "jet_fuel"): 0.6,
|
||||||
|
("fuel_infrastructure", "solid_propellant"): 0.15,
|
||||||
|
("fuel_infrastructure", "nuclear_fuel"): 0.05,
|
||||||
|
("fuel_infrastructure", "xenon_propellant"): 0.05,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Crude freight-capacity proxy: kg of cargo per kg of vehicle structural mass.
|
||||||
|
CARGO_KG_PER_STRUCTURAL_KG: float = 500
|
||||||
|
|
||||||
|
# How mechanically proven/predictable an energy form is in practice — distinct
|
||||||
|
# from safety (risk when something goes wrong) and thrust_profile (delivery
|
||||||
|
# smoothness). Missing values fall back to a neutral 0.6.
|
||||||
|
ENERGY_FORM_RELIABILITY: dict[str, float] = {
|
||||||
|
"chemical_combustible": 0.85,
|
||||||
|
"electrical": 0.85,
|
||||||
|
"biological": 0.8,
|
||||||
|
"gravitational": 0.7,
|
||||||
|
"pneumatic": 0.7,
|
||||||
|
"kinetic_stored": 0.65,
|
||||||
|
"wind": 0.6,
|
||||||
|
"ion_propellant": 0.6,
|
||||||
|
"nuclear_thermal": 0.55,
|
||||||
|
"chemical_propellant": 0.5,
|
||||||
|
"radiation_pressure": 0.5,
|
||||||
|
"chemical_explosive": 0.45,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
@@ -124,7 +230,6 @@ class Pipeline:
|
|||||||
self.repo.update_pipeline_run(run_id, total_combos=len(combos))
|
self.repo.update_pipeline_run(run_id, total_combos=len(combos))
|
||||||
|
|
||||||
# Prepare metric lookup
|
# Prepare metric lookup
|
||||||
metric_names = [mb.metric_name for mb in domain.metric_bounds]
|
|
||||||
bounds_by_name = {mb.metric_name: mb for mb in domain.metric_bounds}
|
bounds_by_name = {mb.metric_name: mb for mb in domain.metric_bounds}
|
||||||
|
|
||||||
# ── Combo-first loop ─────────────────────────────────────
|
# ── Combo-first loop ─────────────────────────────────────
|
||||||
@@ -212,10 +317,10 @@ class Pipeline:
|
|||||||
description = _describe_combination(combo)
|
description = _describe_combination(combo)
|
||||||
if self.llm:
|
if self.llm:
|
||||||
raw_metrics = self.llm.estimate_physics(
|
raw_metrics = self.llm.estimate_physics(
|
||||||
description, metric_names
|
description, domain.metric_bounds
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
raw_metrics = self._stub_estimate(combo, metric_names)
|
raw_metrics = self._stub_estimate(combo, domain.metric_bounds)
|
||||||
|
|
||||||
# Save raw estimates immediately (crash-safe)
|
# Save raw estimates immediately (crash-safe)
|
||||||
estimate_dicts = []
|
estimate_dicts = []
|
||||||
@@ -435,15 +540,32 @@ class Pipeline:
|
|||||||
self.repo.update_pipeline_run(run_id, status="running")
|
self.repo.update_pipeline_run(run_id, status="running")
|
||||||
|
|
||||||
def _stub_estimate(
|
def _stub_estimate(
|
||||||
self, combo: Combination, metric_names: list[str]
|
self, combo: Combination, metric_bounds: list[MetricBound]
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
"""Simple heuristic estimation from dependency data (all values in SI base units)."""
|
"""Simple heuristic estimation from dependency data (all values in SI base units).
|
||||||
|
|
||||||
|
cost_efficiency/safety/availability/reliability are driven by the
|
||||||
|
actuator's thrust_profile and energy_form and the combo's
|
||||||
|
infrastructure requirements — categorical properties every entity
|
||||||
|
already declares — rather than flat constants or a formula that
|
||||||
|
conflates power_density (W/kg, intensive) with cost.
|
||||||
|
|
||||||
|
cost_efficiency additionally checks the domain's declared unit:
|
||||||
|
"$/(kg·m)" (freight-style domains) isn't a rescaling of "$/m" — it's
|
||||||
|
a different quantity that needs dividing by cargo mass, not a
|
||||||
|
conversion factor.
|
||||||
|
"""
|
||||||
|
metric_names = [mb.metric_name for mb in metric_bounds]
|
||||||
|
units_by_name = {mb.metric_name: mb.unit for mb in metric_bounds}
|
||||||
raw: dict[str, float] = {m: 0.0 for m in metric_names}
|
raw: dict[str, float] = {m: 0.0 for m in metric_names}
|
||||||
|
|
||||||
# Extract intrinsic properties from entities
|
# Extract intrinsic properties from entities
|
||||||
power_density = 0.0 # W/kg
|
power_density = 0.0 # W/kg
|
||||||
energy_density = 0.0 # J/kg
|
energy_density = 0.0 # J/kg
|
||||||
mass = 100.0 # kg, default
|
mass_total = 0.0 # kg, extensive — components share one vehicle
|
||||||
|
thrust_profile: str | None = None
|
||||||
|
energy_form: str | None = None
|
||||||
|
infra_matches: list[float] = []
|
||||||
for entity in combo.entities:
|
for entity in combo.entities:
|
||||||
for dep in entity.dependencies:
|
for dep in entity.dependencies:
|
||||||
if dep.key == "power_density" and dep.constraint_type == "provides":
|
if dep.key == "power_density" and dep.constraint_type == "provides":
|
||||||
@@ -451,19 +573,44 @@ class Pipeline:
|
|||||||
if dep.key == "energy_density" and dep.constraint_type == "provides":
|
if dep.key == "energy_density" and dep.constraint_type == "provides":
|
||||||
energy_density = max(energy_density, float(dep.value))
|
energy_density = max(energy_density, float(dep.value))
|
||||||
if dep.key == "mass" and dep.constraint_type == "range_min":
|
if dep.key == "mass" and dep.constraint_type == "range_min":
|
||||||
mass = max(mass, float(dep.value))
|
mass_total += float(dep.value)
|
||||||
|
if dep.key == "thrust_profile" and dep.constraint_type == "provides":
|
||||||
|
thrust_profile = dep.value
|
||||||
|
if dep.key == "energy_form" and dep.constraint_type == "requires":
|
||||||
|
energy_form = dep.value
|
||||||
|
if dep.category == "infrastructure" and dep.constraint_type == "requires":
|
||||||
|
match = INFRASTRUCTURE_AVAILABILITY.get((dep.key, dep.value))
|
||||||
|
if match is not None:
|
||||||
|
infra_matches.append(match)
|
||||||
|
mass = mass_total if mass_total > 0 else 100.0 # kg, default if undeclared
|
||||||
|
cargo_capacity_kg = mass * CARGO_KG_PER_STRUCTURAL_KG
|
||||||
|
|
||||||
if "power_density" in raw:
|
if "power_density" in raw:
|
||||||
raw["power_density"] = power_density
|
raw["power_density"] = power_density
|
||||||
|
|
||||||
if "cost_efficiency" in raw:
|
if "cost_efficiency" in raw:
|
||||||
raw["cost_efficiency"] = max(1e-5, 2e-3 - power_density / 1e6)
|
base_cost = ENERGY_FORM_BASE_COST.get(energy_form, 5e-4)
|
||||||
|
cost_mult = THRUST_PROFILE_COST_MULTIPLIER.get(thrust_profile, 1.0)
|
||||||
|
cost_per_meter = base_cost * cost_mult
|
||||||
|
if units_by_name.get("cost_efficiency") == "$/(kg·m)":
|
||||||
|
raw["cost_efficiency"] = cost_per_meter / max(cargo_capacity_kg, 1.0)
|
||||||
|
else:
|
||||||
|
raw["cost_efficiency"] = cost_per_meter
|
||||||
|
|
||||||
if "safety" in raw:
|
if "safety" in raw:
|
||||||
raw["safety"] = 0.5
|
candidates = [
|
||||||
|
v for v in (
|
||||||
|
THRUST_PROFILE_SAFETY.get(thrust_profile),
|
||||||
|
ENERGY_FORM_SAFETY.get(energy_form),
|
||||||
|
)
|
||||||
|
if v is not None
|
||||||
|
]
|
||||||
|
raw["safety"] = min(candidates) if candidates else 0.6
|
||||||
|
|
||||||
if "availability" in raw:
|
if "availability" in raw:
|
||||||
raw["availability"] = 0.5
|
raw["availability"] = (
|
||||||
|
sum(infra_matches) / len(infra_matches) if infra_matches else 0.5
|
||||||
|
)
|
||||||
|
|
||||||
if "range_fuel" in raw:
|
if "range_fuel" in raw:
|
||||||
raw["range_fuel"] = min(energy_density * 2.78, 1e13)
|
raw["range_fuel"] = min(energy_density * 2.78, 1e13)
|
||||||
@@ -472,7 +619,7 @@ class Pipeline:
|
|||||||
raw["range_degradation"] = 365 * 86400
|
raw["range_degradation"] = 365 * 86400
|
||||||
|
|
||||||
if "cargo_capacity" in raw:
|
if "cargo_capacity" in raw:
|
||||||
raw["cargo_capacity"] = mass * 500
|
raw["cargo_capacity"] = cargo_capacity_kg
|
||||||
|
|
||||||
if "cargo_capacity_kg" in raw:
|
if "cargo_capacity_kg" in raw:
|
||||||
raw["cargo_capacity_kg"] = mass * 0.3
|
raw["cargo_capacity_kg"] = mass * 0.3
|
||||||
@@ -481,6 +628,6 @@ class Pipeline:
|
|||||||
raw["environmental_impact"] = max(0.0, power_density * 2e-7)
|
raw["environmental_impact"] = max(0.0, power_density * 2e-7)
|
||||||
|
|
||||||
if "reliability" in raw:
|
if "reliability" in raw:
|
||||||
raw["reliability"] = 0.5
|
raw["reliability"] = ENERGY_FORM_RELIABILITY.get(energy_form, 0.6)
|
||||||
|
|
||||||
return raw
|
return raw
|
||||||
|
|||||||
@@ -4,6 +4,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
|
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
class LLMRateLimitError(Exception):
|
class LLMRateLimitError(Exception):
|
||||||
"""Raised by a provider when the API rate limit is exceeded.
|
"""Raised by a provider when the API rate limit is exceeded.
|
||||||
@@ -22,10 +24,14 @@ class LLMProvider(ABC):
|
|||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
def estimate_physics(
|
def estimate_physics(
|
||||||
self, combination_description: str, metrics: list[str]
|
self, combination_description: str, metrics: list[MetricBound]
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
"""Given a natural-language description of a combination,
|
"""Given a natural-language description of a combination,
|
||||||
estimate raw metric values. Returns {metric_name: estimated_value}."""
|
estimate raw metric values. `metrics` carries each metric's unit and
|
||||||
|
expected norm_min/norm_max so the estimate lands in the right
|
||||||
|
magnitude — a bare metric name gives no hint that "cost_efficiency"
|
||||||
|
means dollars per meter in the 1e-5 range, not a 0-1 score.
|
||||||
|
Returns {metric_name: estimated_value}."""
|
||||||
...
|
...
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
|
|||||||
30
src/physcom/llm/parsing.py
Normal file
30
src/physcom/llm/parsing.py
Normal file
@@ -0,0 +1,30 @@
|
|||||||
|
"""Shared response-parsing helpers for LLM providers."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
|
def parse_verdict(text: str) -> bool:
|
||||||
|
"""Extract VERDICT: PLAUSIBLE/IMPLAUSIBLE from response; default to True."""
|
||||||
|
m = re.search(r"VERDICT:\s*(PLAUSIBLE|IMPLAUSIBLE)", text, re.IGNORECASE)
|
||||||
|
if m:
|
||||||
|
return m.group(1).upper() == "PLAUSIBLE"
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def parse_metric_json(text: str, metrics: list[MetricBound]) -> dict[str, float]:
|
||||||
|
"""Strip markdown fences and parse JSON; fall back to each metric's own
|
||||||
|
norm_min/norm_max midpoint on error — a flat constant like 0.5 is
|
||||||
|
guaranteed wrong-magnitude for at least some metrics regardless of unit.
|
||||||
|
"""
|
||||||
|
names = {mb.metric_name for mb in metrics}
|
||||||
|
text = re.sub(r"```(?:json)?\s*", "", text).strip().rstrip("`").strip()
|
||||||
|
try:
|
||||||
|
data = json.loads(text)
|
||||||
|
return {k: float(v) for k, v in data.items() if k in names}
|
||||||
|
except (json.JSONDecodeError, ValueError, TypeError):
|
||||||
|
return {mb.metric_name: (mb.norm_min + mb.norm_max) / 2 for mb in metrics}
|
||||||
@@ -1,5 +1,25 @@
|
|||||||
"""Prompt templates for LLM-assisted passes."""
|
"""Prompt templates for LLM-assisted passes."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
|
def format_metrics_for_prompt(metrics: list["MetricBound"]) -> str:
|
||||||
|
"""Render each metric with its unit and expected range, so the model
|
||||||
|
anchors on the right order of magnitude instead of a generic decimal."""
|
||||||
|
lines = []
|
||||||
|
for mb in metrics:
|
||||||
|
unit = mb.unit or "dimensionless"
|
||||||
|
lines.append(
|
||||||
|
f"- {mb.metric_name} ({unit}): typical range {mb.norm_min:g} to {mb.norm_max:g}"
|
||||||
|
)
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
PHYSICS_ESTIMATION_PROMPT = """\
|
PHYSICS_ESTIMATION_PROMPT = """\
|
||||||
You are a physics estimation assistant. Given the following transportation concept, \
|
You are a physics estimation assistant. Given the following transportation concept, \
|
||||||
estimate the requested metrics using order-of-magnitude physics reasoning.
|
estimate the requested metrics using order-of-magnitude physics reasoning.
|
||||||
@@ -8,15 +28,35 @@ estimate the requested metrics using order-of-magnitude physics reasoning.
|
|||||||
{description}
|
{description}
|
||||||
|
|
||||||
## Metrics to estimate
|
## Metrics to estimate
|
||||||
|
Each metric's unit and the typical range values fall in for this domain are given —
|
||||||
|
match that magnitude, don't guess a generically "reasonable-looking" decimal.
|
||||||
{metrics}
|
{metrics}
|
||||||
|
|
||||||
## Instructions
|
## Instructions
|
||||||
- Use real-world physics to estimate each metric.
|
- Use real-world physics to estimate each metric, in the exact unit given.
|
||||||
|
- For "safety" specifically: consider hazards that arise from THIS combination's
|
||||||
|
specific interactions — a fuel that's safe in an open vehicle can be far more
|
||||||
|
dangerous inside a sealed tube or enclosed structure, a stable actuator on a
|
||||||
|
fragile platform can be a real risk even if neither is risky alone. Don't just
|
||||||
|
rate how safe the platform or actuator would be in isolation.
|
||||||
- If the concept is implausible, still provide your best estimate.
|
- If the concept is implausible, still provide your best estimate.
|
||||||
- Return ONLY valid JSON mapping metric names to numeric values.
|
- Return ONLY valid JSON mapping metric names to numeric values, e.g.
|
||||||
- Example: {{"power_density": 500.0, "cost_efficiency": 0.15, "safety": 0.7}}
|
{{"some_metric": <number>, "another_metric": <number>}} — no explanatory text.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
# ponytail: pass 4 only sees pass 2's raw numbers, not its reasoning. Sharpened
|
||||||
|
# prompts on both sides closed most of the gap (a bad safety estimate went from
|
||||||
|
# 0.95 to 0.80 on the same combo once pass 2 was told to consider combination-
|
||||||
|
# specific hazards), but pass 4 still doesn't reliably call out a contradiction
|
||||||
|
# by name when one remains — qwen2.5:7b doesn't follow that meta-instruction
|
||||||
|
# consistently. Upgrade path if this isn't good enough in practice: have
|
||||||
|
# estimate_physics() also return a short per-metric reason, persist it
|
||||||
|
# alongside raw_value (new nullable column), and feed it into this prompt so
|
||||||
|
# pass 4 has something concrete to agree or disagree with. Deferred because it
|
||||||
|
# needs a schema/interface change across LLMProvider + both providers +
|
||||||
|
# pipeline + scorer + repository, and more generated tokens per combo.
|
||||||
|
#
|
||||||
|
# If we plan to LLM-review every p2 pass then maybe p2 and p4 should be combined.
|
||||||
PLAUSIBILITY_REVIEW_PROMPT = """\
|
PLAUSIBILITY_REVIEW_PROMPT = """\
|
||||||
You are reviewing a novel transportation concept for social and practical viability.
|
You are reviewing a novel transportation concept for social and practical viability.
|
||||||
|
|
||||||
@@ -32,6 +72,10 @@ Review this concept for:
|
|||||||
2. Practical barriers — what engineering or regulatory obstacles exist?
|
2. Practical barriers — what engineering or regulatory obstacles exist?
|
||||||
3. Novelty — does anything similar already exist?
|
3. Novelty — does anything similar already exist?
|
||||||
4. Overall plausibility — is this a genuinely interesting innovation or nonsense?
|
4. Overall plausibility — is this a genuinely interesting innovation or nonsense?
|
||||||
|
5. Consistency — if your assessment conflicts with any score above (e.g. you
|
||||||
|
consider this hazardous but its safety score is high), say so explicitly by
|
||||||
|
naming the metric and the discrepancy. Don't silently contradict a given
|
||||||
|
score in your reasoning without calling out that you're doing so.
|
||||||
|
|
||||||
Provide a concise 2-4 sentence assessment, then on a final line write exactly:
|
Provide a concise 2-4 sentence assessment, then on a final line write exactly:
|
||||||
VERDICT: PLAUSIBLE
|
VERDICT: PLAUSIBLE
|
||||||
|
|||||||
@@ -2,12 +2,17 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
|
||||||
import re
|
import re
|
||||||
import math
|
import math
|
||||||
|
|
||||||
from physcom.llm.base import LLMProvider, LLMRateLimitError
|
from physcom.llm.base import LLMProvider, LLMRateLimitError
|
||||||
from physcom.llm.prompts import PHYSICS_ESTIMATION_PROMPT, PLAUSIBILITY_REVIEW_PROMPT
|
from physcom.llm.parsing import parse_metric_json, parse_verdict
|
||||||
|
from physcom.llm.prompts import (
|
||||||
|
PHYSICS_ESTIMATION_PROMPT,
|
||||||
|
PLAUSIBILITY_REVIEW_PROMPT,
|
||||||
|
format_metrics_for_prompt,
|
||||||
|
)
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
class GeminiLLMProvider(LLMProvider):
|
class GeminiLLMProvider(LLMProvider):
|
||||||
@@ -24,11 +29,11 @@ class GeminiLLMProvider(LLMProvider):
|
|||||||
self._model = model
|
self._model = model
|
||||||
|
|
||||||
def estimate_physics(
|
def estimate_physics(
|
||||||
self, combination_description: str, metrics: list[str]
|
self, combination_description: str, metrics: list[MetricBound]
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
prompt = PHYSICS_ESTIMATION_PROMPT.format(
|
prompt = PHYSICS_ESTIMATION_PROMPT.format(
|
||||||
description=combination_description,
|
description=combination_description,
|
||||||
metrics=", ".join(metrics),
|
metrics=format_metrics_for_prompt(metrics),
|
||||||
)
|
)
|
||||||
try:
|
try:
|
||||||
response = self._client.models.generate_content(
|
response = self._client.models.generate_content(
|
||||||
@@ -38,7 +43,7 @@ class GeminiLLMProvider(LLMProvider):
|
|||||||
if "429" in str(exc) or "RESOURCE_EXHAUSTED" in str(exc):
|
if "429" in str(exc) or "RESOURCE_EXHAUSTED" in str(exc):
|
||||||
raise LLMRateLimitError(str(exc), self._parse_retry_after(exc)) from exc
|
raise LLMRateLimitError(str(exc), self._parse_retry_after(exc)) from exc
|
||||||
raise
|
raise
|
||||||
return self._parse_json(response.text, metrics)
|
return parse_metric_json(response.text, metrics)
|
||||||
|
|
||||||
def review_plausibility(
|
def review_plausibility(
|
||||||
self, combination_description: str, scores: dict[str, float]
|
self, combination_description: str, scores: dict[str, float]
|
||||||
@@ -57,26 +62,9 @@ class GeminiLLMProvider(LLMProvider):
|
|||||||
raise LLMRateLimitError(str(exc), self._parse_retry_after(exc)) from exc
|
raise LLMRateLimitError(str(exc), self._parse_retry_after(exc)) from exc
|
||||||
raise
|
raise
|
||||||
text = response.text.strip()
|
text = response.text.strip()
|
||||||
plausible = self._parse_verdict(text)
|
return (text, parse_verdict(text))
|
||||||
return (text, plausible)
|
|
||||||
|
|
||||||
def _parse_verdict(self, text: str) -> bool:
|
|
||||||
"""Extract VERDICT: PLAUSIBLE/IMPLAUSIBLE from response; default to True."""
|
|
||||||
m = re.search(r"VERDICT:\s*(PLAUSIBLE|IMPLAUSIBLE)", text, re.IGNORECASE)
|
|
||||||
if m:
|
|
||||||
return m.group(1).upper() == "PLAUSIBLE"
|
|
||||||
return True
|
|
||||||
|
|
||||||
def _parse_retry_after(self, exc: Exception) -> int:
|
def _parse_retry_after(self, exc: Exception) -> int:
|
||||||
"""Extract retry delay from the error message, with a safe default."""
|
"""Extract retry delay from the error message, with a safe default."""
|
||||||
m = re.search(r"retry in (\d+(?:\.\d+)?)", str(exc))
|
m = re.search(r"retry in (\d+(?:\.\d+)?)", str(exc))
|
||||||
return math.ceil(float(m.group(1))) + 5 if m else 65
|
return math.ceil(float(m.group(1))) + 5 if m else 65
|
||||||
|
|
||||||
def _parse_json(self, text: str, metrics: list[str]) -> dict[str, float]:
|
|
||||||
"""Strip markdown fences and parse JSON; fall back to 0.5 per metric on error."""
|
|
||||||
text = re.sub(r"```(?:json)?\s*", "", text).strip().rstrip("`").strip()
|
|
||||||
try:
|
|
||||||
data = json.loads(text)
|
|
||||||
return {k: float(v) for k, v in data.items() if k in metrics}
|
|
||||||
except (json.JSONDecodeError, ValueError, TypeError):
|
|
||||||
return {m: 0.5 for m in metrics}
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from physcom.llm.base import LLMProvider
|
from physcom.llm.base import LLMProvider
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
class MockLLMProvider(LLMProvider):
|
class MockLLMProvider(LLMProvider):
|
||||||
@@ -12,11 +13,11 @@ class MockLLMProvider(LLMProvider):
|
|||||||
self._defaults = default_estimates or {}
|
self._defaults = default_estimates or {}
|
||||||
|
|
||||||
def estimate_physics(
|
def estimate_physics(
|
||||||
self, combination_description: str, metrics: list[str]
|
self, combination_description: str, metrics: list[MetricBound]
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
result = {}
|
result = {}
|
||||||
for metric in metrics:
|
for mb in metrics:
|
||||||
result[metric] = self._defaults.get(metric, 0.5)
|
result[mb.metric_name] = self._defaults.get(mb.metric_name, 0.5)
|
||||||
return result
|
return result
|
||||||
|
|
||||||
def review_plausibility(
|
def review_plausibility(
|
||||||
|
|||||||
@@ -3,12 +3,17 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import re
|
|
||||||
import urllib.error
|
import urllib.error
|
||||||
import urllib.request
|
import urllib.request
|
||||||
|
|
||||||
from physcom.llm.base import LLMProvider
|
from physcom.llm.base import LLMProvider
|
||||||
from physcom.llm.prompts import PHYSICS_ESTIMATION_PROMPT, PLAUSIBILITY_REVIEW_PROMPT
|
from physcom.llm.parsing import parse_metric_json, parse_verdict
|
||||||
|
from physcom.llm.prompts import (
|
||||||
|
PHYSICS_ESTIMATION_PROMPT,
|
||||||
|
PLAUSIBILITY_REVIEW_PROMPT,
|
||||||
|
format_metrics_for_prompt,
|
||||||
|
)
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
class OllamaLLMProvider(LLMProvider):
|
class OllamaLLMProvider(LLMProvider):
|
||||||
@@ -19,14 +24,14 @@ class OllamaLLMProvider(LLMProvider):
|
|||||||
self._host = host.rstrip("/")
|
self._host = host.rstrip("/")
|
||||||
|
|
||||||
def estimate_physics(
|
def estimate_physics(
|
||||||
self, combination_description: str, metrics: list[str]
|
self, combination_description: str, metrics: list[MetricBound]
|
||||||
) -> dict[str, float]:
|
) -> dict[str, float]:
|
||||||
prompt = PHYSICS_ESTIMATION_PROMPT.format(
|
prompt = PHYSICS_ESTIMATION_PROMPT.format(
|
||||||
description=combination_description,
|
description=combination_description,
|
||||||
metrics=", ".join(metrics),
|
metrics=format_metrics_for_prompt(metrics),
|
||||||
)
|
)
|
||||||
text = self._generate(prompt, json_mode=True)
|
text = self._generate(prompt, json_mode=True)
|
||||||
return self._parse_json(text, metrics)
|
return parse_metric_json(text, metrics)
|
||||||
|
|
||||||
def review_plausibility(
|
def review_plausibility(
|
||||||
self, combination_description: str, scores: dict[str, float]
|
self, combination_description: str, scores: dict[str, float]
|
||||||
@@ -37,7 +42,7 @@ class OllamaLLMProvider(LLMProvider):
|
|||||||
scores=scores_str,
|
scores=scores_str,
|
||||||
)
|
)
|
||||||
text = self._generate(prompt, json_mode=False).strip()
|
text = self._generate(prompt, json_mode=False).strip()
|
||||||
return (text, self._parse_verdict(text))
|
return (text, parse_verdict(text))
|
||||||
|
|
||||||
def _generate(self, prompt: str, json_mode: bool) -> str:
|
def _generate(self, prompt: str, json_mode: bool) -> str:
|
||||||
payload = {"model": self._model, "prompt": prompt, "stream": False}
|
payload = {"model": self._model, "prompt": prompt, "stream": False}
|
||||||
@@ -55,19 +60,3 @@ class OllamaLLMProvider(LLMProvider):
|
|||||||
raise ConnectionError(
|
raise ConnectionError(
|
||||||
f"Could not reach Ollama at {self._host} (is `ollama serve` running?)"
|
f"Could not reach Ollama at {self._host} (is `ollama serve` running?)"
|
||||||
) from exc
|
) from exc
|
||||||
|
|
||||||
def _parse_verdict(self, text: str) -> bool:
|
|
||||||
"""Extract VERDICT: PLAUSIBLE/IMPLAUSIBLE from response; default to True."""
|
|
||||||
m = re.search(r"VERDICT:\s*(PLAUSIBLE|IMPLAUSIBLE)", text, re.IGNORECASE)
|
|
||||||
if m:
|
|
||||||
return m.group(1).upper() == "PLAUSIBLE"
|
|
||||||
return True
|
|
||||||
|
|
||||||
def _parse_json(self, text: str, metrics: list[str]) -> dict[str, float]:
|
|
||||||
"""Strip markdown fences and parse JSON; fall back to 0.5 per metric on error."""
|
|
||||||
text = re.sub(r"```(?:json)?\s*", "", text).strip().rstrip("`").strip()
|
|
||||||
try:
|
|
||||||
data = json.loads(text)
|
|
||||||
return {k: float(v) for k, v in data.items() if k in metrics}
|
|
||||||
except (json.JSONDecodeError, ValueError, TypeError):
|
|
||||||
return {m: 0.5 for m in metrics}
|
|
||||||
|
|||||||
@@ -7,32 +7,40 @@ import os
|
|||||||
from physcom.llm.base import LLMProvider
|
from physcom.llm.base import LLMProvider
|
||||||
|
|
||||||
|
|
||||||
def build_llm_provider() -> LLMProvider | None:
|
def build_llm_provider(
|
||||||
"""Return an LLMProvider based on env vars, or None if not configured.
|
provider: str | None = None,
|
||||||
|
model: str | None = None,
|
||||||
|
host: str | None = None,
|
||||||
|
) -> LLMProvider | None:
|
||||||
|
"""Return an LLMProvider, or None if not configured.
|
||||||
|
|
||||||
|
Explicit args (e.g. from a per-request web form) override env vars;
|
||||||
|
passing nothing falls back to the env-var-only behavior below.
|
||||||
|
|
||||||
LLM_PROVIDER — provider name ('gemini', 'ollama'; more can be added)
|
LLM_PROVIDER — provider name ('gemini', 'ollama'; more can be added)
|
||||||
GEMINI_API_KEY — required when LLM_PROVIDER=gemini
|
GEMINI_API_KEY — required when provider is 'gemini' (server env only,
|
||||||
|
never accepted as a request param)
|
||||||
GEMINI_MODEL — optional Gemini model name (default: gemini-2.0-flash)
|
GEMINI_MODEL — optional Gemini model name (default: gemini-2.0-flash)
|
||||||
OLLAMA_MODEL — optional Ollama model name (default: qwen2.5:7b)
|
OLLAMA_MODEL — optional Ollama model name (default: qwen2.5:7b)
|
||||||
OLLAMA_HOST — optional Ollama server URL (default: http://localhost:11434)
|
OLLAMA_HOST — optional Ollama server URL (default: http://localhost:11434)
|
||||||
"""
|
"""
|
||||||
provider = os.environ.get("LLM_PROVIDER", "").lower().strip()
|
provider = (provider or os.environ.get("LLM_PROVIDER", "")).lower().strip()
|
||||||
|
|
||||||
if not provider:
|
if not provider or provider == "stub":
|
||||||
return None
|
return None
|
||||||
|
|
||||||
if provider == "gemini":
|
if provider == "gemini":
|
||||||
api_key = os.environ.get("GEMINI_API_KEY", "")
|
api_key = os.environ.get("GEMINI_API_KEY", "")
|
||||||
if not api_key:
|
if not api_key:
|
||||||
raise ValueError("LLM_PROVIDER=gemini requires GEMINI_API_KEY to be set")
|
raise ValueError("Gemini requires GEMINI_API_KEY to be set in the server environment")
|
||||||
model = os.environ.get("GEMINI_MODEL", "gemini-2.0-flash")
|
model = model or os.environ.get("GEMINI_MODEL", "gemini-2.0-flash")
|
||||||
from physcom.llm.providers.gemini import GeminiLLMProvider
|
from physcom.llm.providers.gemini import GeminiLLMProvider
|
||||||
return GeminiLLMProvider(api_key=api_key, model=model)
|
return GeminiLLMProvider(api_key=api_key, model=model)
|
||||||
|
|
||||||
if provider == "ollama":
|
if provider == "ollama":
|
||||||
model = os.environ.get("OLLAMA_MODEL", "qwen2.5:7b")
|
model = model or os.environ.get("OLLAMA_MODEL", "qwen2.5:7b")
|
||||||
host = os.environ.get("OLLAMA_HOST", "http://localhost:11434")
|
host = host or os.environ.get("OLLAMA_HOST", "http://localhost:11434")
|
||||||
from physcom.llm.providers.ollama import OllamaLLMProvider
|
from physcom.llm.providers.ollama import OllamaLLMProvider
|
||||||
return OllamaLLMProvider(model=model, host=host)
|
return OllamaLLMProvider(model=model, host=host)
|
||||||
|
|
||||||
raise ValueError(f"Unknown LLM_PROVIDER: {provider!r}. Supported: gemini, ollama")
|
raise ValueError(f"Unknown LLM provider: {provider!r}. Supported: gemini, ollama, stub")
|
||||||
|
|||||||
@@ -15,7 +15,9 @@ GROUND_PLATFORMS: list[Entity] = [
|
|||||||
description="Generic wheeled road vehicle — from motorcycles to trucks",
|
description="Generic wheeled road vehicle — from motorcycles to trucks",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "ground_surface", "true", None, "requires"),
|
Dependency("environment", "ground_surface", "true", None, "requires"),
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "50", "m²", "range_max"),
|
Dependency("physical", "footprint", "50", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "0.5", "m²", "range_min"),
|
Dependency("physical", "footprint", "0.5", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "36000", "kg", "range_max"),
|
Dependency("physical", "mass", "36000", "kg", "range_max"),
|
||||||
@@ -30,7 +32,9 @@ GROUND_PLATFORMS: list[Entity] = [
|
|||||||
description="Small human-scale vehicle — bicycles, skateboards, wheelchairs",
|
description="Small human-scale vehicle — bicycles, skateboards, wheelchairs",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "ground_surface", "true", None, "requires"),
|
Dependency("environment", "ground_surface", "true", None, "requires"),
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "3", "m²", "range_max"),
|
Dependency("physical", "footprint", "3", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "0.3", "m²", "range_min"),
|
Dependency("physical", "footprint", "0.3", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "60", "kg", "range_max"),
|
Dependency("physical", "mass", "60", "kg", "range_max"),
|
||||||
@@ -45,7 +49,9 @@ GROUND_PLATFORMS: list[Entity] = [
|
|||||||
description="Rail-guided vehicle — from trams to high-speed trains",
|
description="Rail-guided vehicle — from trams to high-speed trains",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "ground_surface", "true", None, "requires"),
|
Dependency("environment", "ground_surface", "true", None, "requires"),
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "200", "m²", "range_max"),
|
Dependency("physical", "footprint", "200", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "40000", "kg", "range_max"),
|
Dependency("physical", "mass", "40000", "kg", "range_max"),
|
||||||
@@ -67,6 +73,7 @@ WATER_PLATFORMS: list[Entity] = [
|
|||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "water_surface", "true", None, "requires"),
|
Dependency("environment", "water_surface", "true", None, "requires"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "2000", "m²", "range_max"),
|
Dependency("physical", "footprint", "2000", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "2", "m²", "range_min"),
|
Dependency("physical", "footprint", "2", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "100000", "kg", "range_max"),
|
Dependency("physical", "mass", "100000", "kg", "range_max"),
|
||||||
@@ -81,6 +88,7 @@ WATER_PLATFORMS: list[Entity] = [
|
|||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "water_surface", "true", None, "requires"),
|
Dependency("environment", "water_surface", "true", None, "requires"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "200", "m²", "range_max"),
|
Dependency("physical", "footprint", "200", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "10000", "kg", "range_min"),
|
Dependency("physical", "mass", "10000", "kg", "range_min"),
|
||||||
@@ -101,6 +109,7 @@ AIR_PLATFORMS: list[Entity] = [
|
|||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "500", "m²", "range_max"),
|
Dependency("physical", "footprint", "500", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "10", "m²", "range_min"),
|
Dependency("physical", "footprint", "10", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "100000", "kg", "range_max"),
|
Dependency("physical", "mass", "100000", "kg", "range_max"),
|
||||||
@@ -108,6 +117,7 @@ AIR_PLATFORMS: list[Entity] = [
|
|||||||
Dependency("infrastructure", "runway", "true", None, "requires"),
|
Dependency("infrastructure", "runway", "true", None, "requires"),
|
||||||
Dependency("environment", "medium", "air", None, "requires"),
|
Dependency("environment", "medium", "air", None, "requires"),
|
||||||
Dependency("physical", "energy_density", "1440000", "J/kg", "range_min"),
|
Dependency("physical", "energy_density", "1440000", "J/kg", "range_min"),
|
||||||
|
Dependency("physical", "min_effective_accel", "2.0", "m/s²", "range_min"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
Entity(
|
Entity(
|
||||||
@@ -117,12 +127,14 @@ AIR_PLATFORMS: list[Entity] = [
|
|||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "20", "m²", "range_max"),
|
Dependency("physical", "footprint", "20", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "0.5", "m²", "range_min"),
|
Dependency("physical", "footprint", "0.5", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "5000", "kg", "range_max"),
|
Dependency("physical", "mass", "5000", "kg", "range_max"),
|
||||||
Dependency("physical", "mass", "1", "kg", "range_min"),
|
Dependency("physical", "mass", "1", "kg", "range_min"),
|
||||||
Dependency("environment", "medium", "air", None, "requires"),
|
Dependency("environment", "medium", "air", None, "requires"),
|
||||||
Dependency("physical", "energy_density", "720000", "J/kg", "range_min"),
|
Dependency("physical", "energy_density", "720000", "J/kg", "range_min"),
|
||||||
|
Dependency("physical", "min_effective_accel", "10", "m/s²", "range_min"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
Entity(
|
Entity(
|
||||||
@@ -132,6 +144,7 @@ AIR_PLATFORMS: list[Entity] = [
|
|||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "1000", "m²", "range_max"),
|
Dependency("physical", "footprint", "1000", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "50", "m²", "range_min"),
|
Dependency("physical", "footprint", "50", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "20000", "kg", "range_max"),
|
Dependency("physical", "mass", "20000", "kg", "range_max"),
|
||||||
@@ -139,21 +152,6 @@ AIR_PLATFORMS: list[Entity] = [
|
|||||||
Dependency("environment", "medium", "air", None, "requires"),
|
Dependency("environment", "medium", "air", None, "requires"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
Entity(
|
|
||||||
name="Glider",
|
|
||||||
dimension="platform",
|
|
||||||
description="Unpowered fixed-wing aircraft — sailplanes, hang gliders, paragliders",
|
|
||||||
dependencies=[
|
|
||||||
Dependency("environment", "atmosphere", "standard", None, "requires"),
|
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
|
||||||
Dependency("physical", "footprint", "20", "m²", "range_max"),
|
|
||||||
Dependency("physical", "footprint", "5", "m²", "range_min"),
|
|
||||||
Dependency("physical", "mass", "600", "kg", "range_max"),
|
|
||||||
Dependency("physical", "mass", "5", "kg", "range_min"),
|
|
||||||
Dependency("infrastructure", "tow_or_winch", "true", None, "requires"),
|
|
||||||
Dependency("environment", "medium", "air", None, "requires"),
|
|
||||||
],
|
|
||||||
),
|
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
@@ -172,6 +170,7 @@ SPACE_PLATFORMS: list[Entity] = [
|
|||||||
Dependency("infrastructure", "launch_facility", "true", None, "requires"),
|
Dependency("infrastructure", "launch_facility", "true", None, "requires"),
|
||||||
Dependency("environment", "medium", "space", None, "requires"),
|
Dependency("environment", "medium", "space", None, "requires"),
|
||||||
Dependency("physical", "energy_density", "7200000", "J/kg", "range_min"),
|
Dependency("physical", "energy_density", "7200000", "J/kg", "range_min"),
|
||||||
|
Dependency("physical", "min_effective_accel", "0", "m/s²", "range_min"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
]
|
]
|
||||||
@@ -185,7 +184,9 @@ MULTI_PLATFORMS: list[Entity] = [
|
|||||||
dimension="platform",
|
dimension="platform",
|
||||||
description="Vehicle capable of operation on land, water, or both",
|
description="Vehicle capable of operation on land, water, or both",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "100", "m²", "range_max"),
|
Dependency("physical", "footprint", "100", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "5", "m²", "range_min"),
|
Dependency("physical", "footprint", "5", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "10000", "kg", "range_max"),
|
Dependency("physical", "mass", "10000", "kg", "range_max"),
|
||||||
@@ -198,24 +199,15 @@ MULTI_PLATFORMS: list[Entity] = [
|
|||||||
# ── Platforms — Fictional / Speculative ─────────────────────────
|
# ── Platforms — Fictional / Speculative ─────────────────────────
|
||||||
|
|
||||||
FICTIONAL_PLATFORMS: list[Entity] = [
|
FICTIONAL_PLATFORMS: list[Entity] = [
|
||||||
Entity(
|
|
||||||
name="Teleporter",
|
|
||||||
dimension="platform",
|
|
||||||
description="Hypothetical matter transmission device",
|
|
||||||
dependencies=[
|
|
||||||
Dependency("physical", "footprint", "10", "m²", "range_max"),
|
|
||||||
Dependency("physical", "footprint", "1", "m²", "range_min"),
|
|
||||||
Dependency("physical", "mass", "0", "kg", "range_min"),
|
|
||||||
Dependency("infrastructure", "teleport_network", "true", None, "requires"),
|
|
||||||
],
|
|
||||||
),
|
|
||||||
Entity(
|
Entity(
|
||||||
name="Hyperloop",
|
name="Hyperloop",
|
||||||
dimension="platform",
|
dimension="platform",
|
||||||
description="Sealed low-pressure tube with passenger pods at near-sonic speed",
|
description="Sealed low-pressure tube with passenger pods at near-sonic speed",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "ground_surface", "true", None, "requires"),
|
Dependency("environment", "ground_surface", "true", None, "requires"),
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "footprint", "50", "m²", "range_max"),
|
Dependency("physical", "footprint", "50", "m²", "range_max"),
|
||||||
Dependency("physical", "footprint", "5", "m²", "range_min"),
|
Dependency("physical", "footprint", "5", "m²", "range_min"),
|
||||||
Dependency("physical", "mass", "20000", "kg", "range_max"),
|
Dependency("physical", "mass", "20000", "kg", "range_max"),
|
||||||
@@ -264,6 +256,7 @@ COMBUSTION_ACTUATORS: list[Entity] = [
|
|||||||
Dependency("physical", "mass", "200", "kg", "range_min"),
|
Dependency("physical", "mass", "200", "kg", "range_min"),
|
||||||
Dependency("force", "thrust_profile", "extreme_continuous", None, "provides"),
|
Dependency("force", "thrust_profile", "extreme_continuous", None, "provides"),
|
||||||
Dependency("force", "power_density", "5000", "W/kg", "provides"),
|
Dependency("force", "power_density", "5000", "W/kg", "provides"),
|
||||||
|
Dependency("force", "specific_thrust", "50", "N/kg", "provides"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
Entity(
|
Entity(
|
||||||
@@ -368,9 +361,11 @@ ROCKET_ACTUATORS: list[Entity] = [
|
|||||||
description="Thrust from expanding combustion gases through a nozzle",
|
description="Thrust from expanding combustion gases through a nozzle",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("energy", "energy_form", "chemical_propellant", None, "requires"),
|
Dependency("energy", "energy_form", "chemical_propellant", None, "requires"),
|
||||||
|
Dependency("environment", "medium", "water", None, "excludes"),
|
||||||
Dependency("physical", "mass", "150", "kg", "range_min"),
|
Dependency("physical", "mass", "150", "kg", "range_min"),
|
||||||
Dependency("force", "thrust_profile", "extreme_burst", None, "provides"),
|
Dependency("force", "thrust_profile", "extreme_burst", None, "provides"),
|
||||||
Dependency("force", "power_density", "10000", "W/kg", "provides"),
|
Dependency("force", "power_density", "10000", "W/kg", "provides"),
|
||||||
|
Dependency("force", "specific_thrust", "1500", "N/kg", "provides"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
Entity(
|
Entity(
|
||||||
@@ -384,6 +379,7 @@ ROCKET_ACTUATORS: list[Entity] = [
|
|||||||
Dependency("physical", "mass", "8", "kg", "range_min"),
|
Dependency("physical", "mass", "8", "kg", "range_min"),
|
||||||
Dependency("force", "thrust_profile", "continuous_low", None, "provides"),
|
Dependency("force", "thrust_profile", "continuous_low", None, "provides"),
|
||||||
Dependency("force", "power_density", "30", "W/kg", "provides"),
|
Dependency("force", "power_density", "30", "W/kg", "provides"),
|
||||||
|
Dependency("force", "specific_thrust", "0.01", "N/kg", "provides"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
Entity(
|
Entity(
|
||||||
@@ -396,6 +392,7 @@ ROCKET_ACTUATORS: list[Entity] = [
|
|||||||
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
||||||
Dependency("force", "thrust_profile", "extreme_continuous", None, "provides"),
|
Dependency("force", "thrust_profile", "extreme_continuous", None, "provides"),
|
||||||
Dependency("force", "power_density", "50", "W/kg", "provides"),
|
Dependency("force", "power_density", "50", "W/kg", "provides"),
|
||||||
|
Dependency("force", "specific_thrust", "200", "N/kg", "provides"),
|
||||||
Dependency("material", "radiation_shielding", "true", None, "requires"),
|
Dependency("material", "radiation_shielding", "true", None, "requires"),
|
||||||
],
|
],
|
||||||
),
|
),
|
||||||
@@ -411,6 +408,7 @@ EXOTIC_ACTUATORS: list[Entity] = [
|
|||||||
description="Propulsion via sequential cannon blasts",
|
description="Propulsion via sequential cannon blasts",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("energy", "energy_form", "chemical_explosive", None, "requires"),
|
Dependency("energy", "energy_form", "chemical_explosive", None, "requires"),
|
||||||
|
Dependency("environment", "medium", "water", None, "excludes"),
|
||||||
Dependency("physical", "mass", "80", "kg", "range_min"),
|
Dependency("physical", "mass", "80", "kg", "range_min"),
|
||||||
Dependency("force", "thrust_profile", "high_burst", None, "provides"),
|
Dependency("force", "thrust_profile", "high_burst", None, "provides"),
|
||||||
Dependency("force", "power_density", "3000", "W/kg", "provides"),
|
Dependency("force", "power_density", "3000", "W/kg", "provides"),
|
||||||
@@ -836,6 +834,7 @@ def load_transport_seed(repo) -> dict:
|
|||||||
# Backfill metric units and lower_is_better on existing DBs.
|
# Backfill metric units and lower_is_better on existing DBs.
|
||||||
for mb in domain.metric_bounds:
|
for mb in domain.metric_bounds:
|
||||||
repo.ensure_metric(mb.metric_name, unit=mb.unit)
|
repo.ensure_metric(mb.metric_name, unit=mb.unit)
|
||||||
|
repo.backfill_metric_unit(domain.name, mb.metric_name, mb.unit)
|
||||||
if mb.lower_is_better:
|
if mb.lower_is_better:
|
||||||
repo.backfill_lower_is_better(domain.name, mb.metric_name)
|
repo.backfill_lower_is_better(domain.name, mb.metric_name)
|
||||||
# Backfill domain constraints
|
# Backfill domain constraints
|
||||||
|
|||||||
@@ -21,6 +21,9 @@ def _run_pipeline_in_background(
|
|||||||
passes: list[int],
|
passes: list[int],
|
||||||
threshold: float,
|
threshold: float,
|
||||||
run_id: int,
|
run_id: int,
|
||||||
|
llm_provider: str | None = None,
|
||||||
|
llm_model: str | None = None,
|
||||||
|
llm_host: str | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Run the pipeline in a background thread with its own DB connection."""
|
"""Run the pipeline in a background thread with its own DB connection."""
|
||||||
from physcom.db.schema import init_db
|
from physcom.db.schema import init_db
|
||||||
@@ -45,7 +48,8 @@ def _run_pipeline_in_background(
|
|||||||
from physcom.llm.registry import build_llm_provider
|
from physcom.llm.registry import build_llm_provider
|
||||||
resolver = ConstraintResolver()
|
resolver = ConstraintResolver()
|
||||||
scorer = Scorer(domain)
|
scorer = Scorer(domain)
|
||||||
pipeline = Pipeline(repo, resolver, scorer, llm=build_llm_provider())
|
llm = build_llm_provider(provider=llm_provider, model=llm_model, host=llm_host)
|
||||||
|
pipeline = Pipeline(repo, resolver, scorer, llm=llm)
|
||||||
|
|
||||||
pipeline.run(
|
pipeline.run(
|
||||||
domain, dim_list,
|
domain, dim_list,
|
||||||
@@ -108,11 +112,17 @@ def pipeline_run():
|
|||||||
flash("Select at least one dimension.", "error")
|
flash("Select at least one dimension.", "error")
|
||||||
return redirect(url_for("pipeline.pipeline_form"))
|
return redirect(url_for("pipeline.pipeline_form"))
|
||||||
|
|
||||||
|
llm_provider = request.form.get("llm_provider", "").strip() or None
|
||||||
|
llm_model = request.form.get("llm_model", "").strip() or None
|
||||||
|
llm_host = request.form.get("llm_host", "").strip() or None
|
||||||
|
|
||||||
# Create pipeline_run record
|
# Create pipeline_run record
|
||||||
config = {
|
config = {
|
||||||
"passes": passes,
|
"passes": passes,
|
||||||
"threshold": threshold,
|
"threshold": threshold,
|
||||||
"dimensions": dim_list,
|
"dimensions": dim_list,
|
||||||
|
"llm_provider": llm_provider,
|
||||||
|
"llm_model": llm_model,
|
||||||
}
|
}
|
||||||
run_id = repo.create_pipeline_run(domain.id, config)
|
run_id = repo.create_pipeline_run(domain.id, config)
|
||||||
|
|
||||||
@@ -123,7 +133,8 @@ def pipeline_run():
|
|||||||
# Start background thread
|
# Start background thread
|
||||||
t = threading.Thread(
|
t = threading.Thread(
|
||||||
target=_run_pipeline_in_background,
|
target=_run_pipeline_in_background,
|
||||||
args=(db_path, domain_name, dim_list, passes, threshold, run_id),
|
args=(db_path, domain_name, dim_list, passes, threshold, run_id,
|
||||||
|
llm_provider, llm_model, llm_host),
|
||||||
daemon=True,
|
daemon=True,
|
||||||
)
|
)
|
||||||
t.start()
|
t.start()
|
||||||
|
|||||||
@@ -49,6 +49,29 @@
|
|||||||
</div>
|
</div>
|
||||||
</fieldset>
|
</fieldset>
|
||||||
|
|
||||||
|
<fieldset>
|
||||||
|
<legend>LLM Provider</legend>
|
||||||
|
<p class="form-hint">Used for Pass 2 estimation and Pass 4 review. Leave on "server default" to use whatever LLM_PROVIDER is configured in the server environment (or the physics stub if none).</p>
|
||||||
|
<div class="form-group">
|
||||||
|
<select name="llm_provider" id="llm_provider">
|
||||||
|
<option value="">— server default —</option>
|
||||||
|
<option value="stub">Stub (fast, no LLM)</option>
|
||||||
|
<option value="ollama">Ollama (local)</option>
|
||||||
|
<option value="gemini">Gemini (cloud, requires server-side GEMINI_API_KEY)</option>
|
||||||
|
</select>
|
||||||
|
</div>
|
||||||
|
<div class="form-group">
|
||||||
|
<label for="llm_model">Model</label>
|
||||||
|
<p class="form-hint">Leave blank to use the provider's default model.</p>
|
||||||
|
<input type="text" name="llm_model" id="llm_model" placeholder="e.g. qwen2.5:7b or gemini-2.0-flash">
|
||||||
|
</div>
|
||||||
|
<div class="form-group">
|
||||||
|
<label for="llm_host">Ollama host</label>
|
||||||
|
<p class="form-hint">Only used when Ollama is selected. Leave blank for http://localhost:11434.</p>
|
||||||
|
<input type="text" name="llm_host" id="llm_host" placeholder="http://localhost:11434">
|
||||||
|
</div>
|
||||||
|
</fieldset>
|
||||||
|
|
||||||
<div class="form-group">
|
<div class="form-group">
|
||||||
<label for="threshold">Score Threshold</label>
|
<label for="threshold">Score Threshold</label>
|
||||||
<p class="form-hint">Minimum composite score (0–1) for a combination to pass scoring. Lower values keep more results; higher values are more selective.</p>
|
<p class="form-hint">Minimum composite score (0–1) for a combination to pass scoring. Lower values keep more results; higher values are more selective.</p>
|
||||||
|
|||||||
@@ -35,7 +35,9 @@ def road_vehicle():
|
|||||||
description="Generic wheeled road vehicle",
|
description="Generic wheeled road vehicle",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "ground_surface", "true", None, "requires"),
|
Dependency("environment", "ground_surface", "true", None, "requires"),
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "mass", "36000", "kg", "range_max"),
|
Dependency("physical", "mass", "36000", "kg", "range_max"),
|
||||||
Dependency("physical", "mass", "50", "kg", "range_min"),
|
Dependency("physical", "mass", "50", "kg", "range_min"),
|
||||||
Dependency("environment", "medium", "ground", None, "requires"),
|
Dependency("environment", "medium", "ground", None, "requires"),
|
||||||
@@ -51,7 +53,9 @@ def bicycle():
|
|||||||
description="Two-wheeled human-scale vehicle",
|
description="Two-wheeled human-scale vehicle",
|
||||||
dependencies=[
|
dependencies=[
|
||||||
Dependency("environment", "ground_surface", "true", None, "requires"),
|
Dependency("environment", "ground_surface", "true", None, "requires"),
|
||||||
|
Dependency("environment", "ground_surface", "true", None, "provides"),
|
||||||
Dependency("environment", "gravity", "true", None, "requires"),
|
Dependency("environment", "gravity", "true", None, "requires"),
|
||||||
|
Dependency("environment", "gravity", "true", None, "provides"),
|
||||||
Dependency("physical", "mass", "30", "kg", "range_max"),
|
Dependency("physical", "mass", "30", "kg", "range_max"),
|
||||||
Dependency("environment", "medium", "ground", None, "requires"),
|
Dependency("environment", "medium", "ground", None, "requires"),
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -160,3 +160,231 @@ def test_domain_constraint_allows_matching_medium(bicycle, human_pedalling, food
|
|||||||
constraints = [DomainConstraint("medium", ["ground", "air"])]
|
constraints = [DomainConstraint("medium", ["ground", "air"])]
|
||||||
result = resolver.check_domain_constraints(combo, constraints)
|
result = resolver.check_domain_constraints(combo, constraints)
|
||||||
assert result.status == "valid"
|
assert result.status == "valid"
|
||||||
|
|
||||||
|
|
||||||
|
def _rotorcraft():
|
||||||
|
return Entity(
|
||||||
|
name="Rotorcraft", dimension="platform",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "footprint", "20", "m²", "range_max"),
|
||||||
|
Dependency("physical", "footprint", "0.5", "m²", "range_min"),
|
||||||
|
Dependency("physical", "mass", "5000", "kg", "range_max"),
|
||||||
|
Dependency("physical", "mass", "1", "kg", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _spaceship_with_footprint():
|
||||||
|
return Entity(
|
||||||
|
name="Spaceship", dimension="platform",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "footprint", "500", "m²", "range_max"),
|
||||||
|
Dependency("physical", "footprint", "10", "m²", "range_min"),
|
||||||
|
Dependency("physical", "mass", "5000", "kg", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _nuclear_thermal_drive_with_footprint():
|
||||||
|
return Entity(
|
||||||
|
name="Nuclear Thermal Drive", dimension="actuator",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "footprint", "20", "m²", "range_min"),
|
||||||
|
Dependency("physical", "mass", "1500", "kg", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _nuclear_fuel_with_footprint():
|
||||||
|
return Entity(
|
||||||
|
name="Nuclear Fuel", dimension="energy_storage",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "footprint", "5", "m²", "range_min"),
|
||||||
|
Dependency("physical", "mass", "500", "kg", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_footprint_aggregation_blocks_reactor_on_rotorcraft():
|
||||||
|
"""P1: individual footprint floors each fit under the ceiling (20, 5 <= 20)
|
||||||
|
but their sum (25.5) doesn't — must block even though no single component
|
||||||
|
exceeds the ceiling on its own."""
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
combo = Combination(entities=[
|
||||||
|
_rotorcraft(), _nuclear_thermal_drive_with_footprint(), _nuclear_fuel_with_footprint(),
|
||||||
|
])
|
||||||
|
result = resolver.resolve(combo)
|
||||||
|
assert result.status == "p1_fail"
|
||||||
|
assert any("combined footprint" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_footprint_aggregation_still_passes_spaceship():
|
||||||
|
"""Same reactor + fuel, but a platform with enough footprint budget (500 m²)
|
||||||
|
must still pass — aggregation shouldn't over-block combos with real headroom."""
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
combo = Combination(entities=[
|
||||||
|
_spaceship_with_footprint(), _nuclear_thermal_drive_with_footprint(), _nuclear_fuel_with_footprint(),
|
||||||
|
])
|
||||||
|
result = resolver.resolve(combo)
|
||||||
|
assert result.status != "p1_fail"
|
||||||
|
assert not any("footprint" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_mass_aggregation_within_tolerance_warns_not_blocks():
|
||||||
|
"""Sum only slightly over the ceiling (65 vs 60, +8.3%) is a data-calibration
|
||||||
|
signal, not a hard physical impossibility — should warn, not block."""
|
||||||
|
platform = Entity(
|
||||||
|
name="Light Personal Vehicle", dimension="platform",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "60", "kg", "range_max"),
|
||||||
|
Dependency("physical", "mass", "5", "kg", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
actuator = Entity(
|
||||||
|
name="Piston Engine", dimension="actuator",
|
||||||
|
dependencies=[Dependency("physical", "mass", "45", "kg", "range_min")],
|
||||||
|
)
|
||||||
|
storage = Entity(
|
||||||
|
name="Compressed Natural Gas", dimension="energy_storage",
|
||||||
|
dependencies=[Dependency("physical", "mass", "15", "kg", "range_min")],
|
||||||
|
)
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[platform, actuator, storage]))
|
||||||
|
assert result.status == "conditional"
|
||||||
|
assert any("combined mass" in w for w in result.warnings)
|
||||||
|
|
||||||
|
|
||||||
|
def test_weak_secondary_provider_does_not_block_satisfied_requirement():
|
||||||
|
"""P3: a strong provider (nuclear fuel) already satisfies the requirement;
|
||||||
|
a weak secondary provider (solar panel) in the same combo must not
|
||||||
|
retroactively block it — a real backup power source shouldn't break a
|
||||||
|
vehicle that already has enough primary power."""
|
||||||
|
platform = Entity(
|
||||||
|
name="Spaceship", dimension="platform",
|
||||||
|
dependencies=[Dependency("physical", "energy_density", "7200000", "J/kg", "range_min")],
|
||||||
|
)
|
||||||
|
nuclear_fuel = Entity(
|
||||||
|
name="Nuclear Fuel", dimension="energy_storage",
|
||||||
|
dependencies=[Dependency("physical", "energy_density", "1800000000", "J/kg", "provides")],
|
||||||
|
)
|
||||||
|
solar_panel = Entity(
|
||||||
|
name="Solar Photovoltaic Panel", dimension="energy_storage",
|
||||||
|
dependencies=[Dependency("physical", "energy_density", "180000", "J/kg", "provides")],
|
||||||
|
)
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[platform, nuclear_fuel, solar_panel]))
|
||||||
|
assert result.status != "p1_fail"
|
||||||
|
assert not any("energy_density" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_unrecognized_mutex_value_fails_closed():
|
||||||
|
"""P4: a value not in any registered mutex set (e.g. a new 'medium' typed
|
||||||
|
into the admin UI) must conflict with a recognized value on the same key,
|
||||||
|
not silently pass."""
|
||||||
|
a = Entity(
|
||||||
|
name="A", dimension="platform",
|
||||||
|
dependencies=[Dependency("environment", "medium", "underground", None, "requires")],
|
||||||
|
)
|
||||||
|
b = Entity(
|
||||||
|
name="B", dimension="actuator",
|
||||||
|
dependencies=[Dependency("environment", "medium", "space", None, "requires")],
|
||||||
|
)
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[a, b]))
|
||||||
|
assert result.status == "p1_fail"
|
||||||
|
assert any("mutually exclusive" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_propulsion_viability_blocks_weak_actuator_regardless_of_scale():
|
||||||
|
"""G4: specific_thrust below min_effective_accel can never be fixed by
|
||||||
|
adding more actuator mass — must block unconditionally (Case 1)."""
|
||||||
|
platform = Entity(
|
||||||
|
name="Rotorcraft", dimension="platform",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "5000", "kg", "range_max"),
|
||||||
|
Dependency("physical", "min_effective_accel", "10", "m/s²", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
actuator = Entity(
|
||||||
|
name="Ion Drive", dimension="actuator",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "8", "kg", "range_min"),
|
||||||
|
Dependency("force", "specific_thrust", "0.01", "N/kg", "provides"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[platform, actuator]))
|
||||||
|
assert result.status == "p1_fail"
|
||||||
|
assert any("regardless of scale" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_propulsion_viability_blocks_when_required_mass_exceeds_ceiling():
|
||||||
|
"""G4 Case 2: specific_thrust clears min_effective_accel, but the mass
|
||||||
|
needed to hit that thrust doesn't fit the vehicle's mass budget."""
|
||||||
|
platform = Entity(
|
||||||
|
name="Test Platform", dimension="platform",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "50", "kg", "range_max"),
|
||||||
|
Dependency("physical", "mass", "10", "kg", "range_min"),
|
||||||
|
Dependency("physical", "min_effective_accel", "5", "m/s²", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
actuator = Entity(
|
||||||
|
name="Weak Reaction Drive", dimension="actuator",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "1", "kg", "range_min"),
|
||||||
|
Dependency("force", "specific_thrust", "6", "N/kg", "provides"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
storage = Entity(
|
||||||
|
name="Fuel", dimension="energy_storage",
|
||||||
|
dependencies=[Dependency("physical", "mass", "1", "kg", "range_min")],
|
||||||
|
)
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[platform, actuator, storage]))
|
||||||
|
assert result.status == "p1_fail"
|
||||||
|
assert any("would need >=" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_propulsion_viability_passes_with_enough_budget():
|
||||||
|
"""Same shape as above but with a generous mass ceiling — must pass."""
|
||||||
|
platform = Entity(
|
||||||
|
name="Test Platform", dimension="platform",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "5000", "kg", "range_max"),
|
||||||
|
Dependency("physical", "mass", "10", "kg", "range_min"),
|
||||||
|
Dependency("physical", "min_effective_accel", "5", "m/s²", "range_min"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
actuator = Entity(
|
||||||
|
name="Weak Reaction Drive", dimension="actuator",
|
||||||
|
dependencies=[
|
||||||
|
Dependency("physical", "mass", "1", "kg", "range_min"),
|
||||||
|
Dependency("force", "specific_thrust", "6", "N/kg", "provides"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
storage = Entity(
|
||||||
|
name="Fuel", dimension="energy_storage",
|
||||||
|
dependencies=[Dependency("physical", "mass", "1", "kg", "range_min")],
|
||||||
|
)
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[platform, actuator, storage]))
|
||||||
|
assert not any("specific thrust" in v or "would need" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_propulsion_viability_skips_when_undeclared(bicycle, human_pedalling, food_calories):
|
||||||
|
"""Platforms/actuators that never declare min_effective_accel or
|
||||||
|
specific_thrust (most of the catalog, for now) must be unaffected."""
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[bicycle, human_pedalling, food_calories]))
|
||||||
|
assert not any("specific thrust" in v or "would need" in v for v in result.violations)
|
||||||
|
|
||||||
|
|
||||||
|
def test_agreement_key_reaches_valid_status(bicycle, human_pedalling, food_calories):
|
||||||
|
"""P2: medium/atmosphere are agreement keys, not supply/demand — a combo
|
||||||
|
with no other issues should reach 'valid', not get stuck at 'conditional'
|
||||||
|
forever because nothing 'provides' medium=ground."""
|
||||||
|
resolver = ConstraintResolver()
|
||||||
|
result = resolver.resolve(Combination(entities=[bicycle, human_pedalling, food_calories]))
|
||||||
|
assert result.status == "valid"
|
||||||
|
assert not any("medium" in w for w in result.warnings)
|
||||||
|
|||||||
32
tests/test_llm_parsing.py
Normal file
32
tests/test_llm_parsing.py
Normal file
@@ -0,0 +1,32 @@
|
|||||||
|
"""Tests for shared LLM response-parsing logic."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from physcom.llm.parsing import parse_metric_json, parse_verdict
|
||||||
|
from physcom.models.domain import MetricBound
|
||||||
|
|
||||||
|
|
||||||
|
def _bounds():
|
||||||
|
return [
|
||||||
|
MetricBound("power_density", weight=0.5, norm_min=1, norm_max=2000, unit="W/kg"),
|
||||||
|
MetricBound("safety", weight=0.5, norm_min=0.0, norm_max=1.0, unit="0-1"),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_metric_json_strips_fences():
|
||||||
|
text = '```json\n{"power_density": 500.0, "safety": 0.7}\n```'
|
||||||
|
result = parse_metric_json(text, _bounds())
|
||||||
|
assert result == {"power_density": 500.0, "safety": 0.7}
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_metric_json_falls_back_to_range_midpoint_on_invalid():
|
||||||
|
result = parse_metric_json("not json", _bounds())
|
||||||
|
assert result == {"power_density": 1000.5, "safety": 0.5}
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_verdict_plausible():
|
||||||
|
assert parse_verdict("blah blah\nVERDICT: PLAUSIBLE") is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_verdict_implausible():
|
||||||
|
assert parse_verdict("blah blah\nVERDICT: IMPLAUSIBLE") is False
|
||||||
@@ -1,36 +1,10 @@
|
|||||||
"""Tests for the Ollama provider's parsing logic and registry wiring."""
|
"""Tests for the Ollama provider's registry wiring."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from physcom.llm.providers.ollama import OllamaLLMProvider
|
from physcom.llm.providers.ollama import OllamaLLMProvider
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
|
||||||
def provider():
|
|
||||||
return OllamaLLMProvider()
|
|
||||||
|
|
||||||
|
|
||||||
def test_parse_json_strips_fences(provider):
|
|
||||||
text = '```json\n{"power_density": 500.0, "safety": 0.7}\n```'
|
|
||||||
result = provider._parse_json(text, ["power_density", "safety"])
|
|
||||||
assert result == {"power_density": 500.0, "safety": 0.7}
|
|
||||||
|
|
||||||
|
|
||||||
def test_parse_json_falls_back_on_invalid(provider):
|
|
||||||
result = provider._parse_json("not json", ["power_density", "safety"])
|
|
||||||
assert result == {"power_density": 0.5, "safety": 0.5}
|
|
||||||
|
|
||||||
|
|
||||||
def test_parse_verdict_plausible(provider):
|
|
||||||
assert provider._parse_verdict("blah blah\nVERDICT: PLAUSIBLE") is True
|
|
||||||
|
|
||||||
|
|
||||||
def test_parse_verdict_implausible(provider):
|
|
||||||
assert provider._parse_verdict("blah blah\nVERDICT: IMPLAUSIBLE") is False
|
|
||||||
|
|
||||||
|
|
||||||
def test_registry_builds_ollama_provider(monkeypatch):
|
def test_registry_builds_ollama_provider(monkeypatch):
|
||||||
from physcom.llm.registry import build_llm_provider
|
from physcom.llm.registry import build_llm_provider
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user