give pass 4 raw values, batch deferrable commits, fix two known bugs
Pass 4's plausibility review only ever saw normalized scores, never the raw physical estimate behind them -- confirmed via live testing this was exactly what caused a real misfire (gemma2:27b cited a real cyclist's correct 5 W/kg, log-normalized to "0.159" against a car's power scale, as grounds for rejecting an ordinary bicycle). review_plausibility now takes raw_metrics + normalized_scores + metric units, and the prompt explicitly instructs reasoning from the raw value first. Verified live against phi4: it now cites the actual raw number and correctly explains why a low normalized score doesn't mean the estimate or concept is bad. Repository write methods used in the pipeline's hot path now take an optional commit=False, and Pipeline defers commits during the fast/ deterministic passes (1, 3, and 2 without an LLM), flushing every 200 combos and on any exit path (finally block covers normal completion, cancellation, and any other exception). LLM-involving calls (pass 2 with an LLM, all of pass 4) still commit immediately -- those are slow and crash-prone and worth protecting per-write; the deterministic passes aren't, and recomputing them is now measured at under a second for the full domain rather than worth 8,000+ individual fsync'd commits. Full 2,970-combination domain run: multiple minutes -> 0.91s. Test suite: ~70s -> ~15s. Also fixes two more issues found while auditing the estimator for a real run: CARGO_KG_PER_STRUCTURAL_KG was 500 (no real vehicle carries 500x its own structural mass in cargo -- a magnitude bug, not a modeling choice), corrected to 2.5. And space/rocket platforms' range_fuel now reports the domain's ceiling instead of an arbitrary placeholder constant -- vacuum coast isn't resistance-limited, so "distance before running out of fuel" isn't a meaningful question for these the way it is for ground/air/water vehicles; the real constraint is delta-v budget, a different metric this pass doesn't model. Validated with a live full-domain run (phi4, real Ollama calls): 115 reviewed, 0 malformed/null reviews, 0 verdict-vs-status mismatches. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -36,8 +36,20 @@ class LLMProvider(ABC):
|
||||
|
||||
@abstractmethod
|
||||
def review_plausibility(
|
||||
self, combination_description: str, scores: dict[str, float]
|
||||
self,
|
||||
combination_description: str,
|
||||
raw_metrics: dict[str, float],
|
||||
normalized_scores: dict[str, float],
|
||||
metrics: list[MetricBound],
|
||||
) -> tuple[str, bool]:
|
||||
"""Given a combination and its scores, return a (text, is_plausible)
|
||||
tuple: natural-language assessment and whether the concept is plausible."""
|
||||
"""Given a combination, its raw physical estimates, and their
|
||||
normalized scores, return a (text, is_plausible) tuple:
|
||||
natural-language assessment and whether the concept is plausible.
|
||||
|
||||
Both raw_metrics and normalized_scores are given (not just the
|
||||
normalized score) so the review can reason from the actual physics
|
||||
rather than only a compressed 0-1 number, which can look
|
||||
deceptively bad for a metric whose scale was built for a different
|
||||
kind of vehicle. `metrics` carries each metric's unit for
|
||||
formatting the raw value meaningfully."""
|
||||
...
|
||||
|
||||
@@ -20,6 +20,35 @@ def format_metrics_for_prompt(metrics: list["MetricBound"]) -> str:
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def format_scores_for_prompt(
|
||||
raw_metrics: dict[str, float],
|
||||
normalized_scores: dict[str, float],
|
||||
metrics: list["MetricBound"],
|
||||
) -> str:
|
||||
"""Render each metric with BOTH its raw physical value and its
|
||||
normalized score, so the reviewing pass can reason from the actual
|
||||
physics instead of only ever seeing a compressed 0-1 number.
|
||||
|
||||
A real, correct estimate can still look damning once log-normalized
|
||||
against a scale built for a different kind of vehicle (a cyclist's
|
||||
real ~5 W/kg reads as "0.159" next to a car's 2000 W/kg ceiling) --
|
||||
a reviewer that only sees the 0.159 has no way to notice that. See
|
||||
the labeled-set calibration note on PLAUSIBILITY_REVIEW_PROMPT below.
|
||||
"""
|
||||
lines = []
|
||||
for mb in metrics:
|
||||
normed = normalized_scores.get(mb.metric_name)
|
||||
if normed is None:
|
||||
continue
|
||||
raw = raw_metrics.get(mb.metric_name)
|
||||
unit = mb.unit or "dimensionless"
|
||||
raw_str = f"{raw:g} {unit}" if raw is not None else "unknown"
|
||||
lines.append(
|
||||
f"- {mb.metric_name}: raw estimate {raw_str} — normalized score {normed:.3f}"
|
||||
)
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
PHYSICS_ESTIMATION_PROMPT = """\
|
||||
You are a physics estimation assistant. Given the following transportation concept, \
|
||||
estimate the requested metrics using order-of-magnitude physics reasoning.
|
||||
@@ -44,15 +73,22 @@ match that magnitude, don't guess a generically "reasonable-looking" decimal.
|
||||
{{"some_metric": <number>, "another_metric": <number>}} — no explanatory text.
|
||||
"""
|
||||
|
||||
# ponytail: pass 4 only sees pass 2's raw numbers, not its reasoning. Sharpened
|
||||
# prompts on both sides closed most of the gap (a bad safety estimate went from
|
||||
# 0.95 to 0.80 on the same combo once pass 2 was told to consider combination-
|
||||
# specific hazards). Upgrade path if this isn't good enough in practice: have
|
||||
# estimate_physics() also return a short per-metric reason, persist it
|
||||
# alongside raw_value (new nullable column), and feed it into this prompt so
|
||||
# pass 4 has something concrete to agree or disagree with. Deferred because it
|
||||
# needs a schema/interface change across LLMProvider + both providers +
|
||||
# pipeline + scorer + repository, and more generated tokens per combo.
|
||||
# ponytail: pass 4 used to see only pass 2's normalized scores, not the raw
|
||||
# physical numbers or any reasoning behind them. Fixed the raw-value half of
|
||||
# that gap: format_scores_for_prompt() now shows both, since a correct raw
|
||||
# estimate can look damning once log-normalized against a scale built for a
|
||||
# different kind of vehicle (a cyclist's real ~5 W/kg reads as "0.159" next
|
||||
# to a car's 2000 W/kg ceiling) -- gemma2:27b did exactly this on a real
|
||||
# bicycle combo, citing "extremely low power density (0.159)" as grounds for
|
||||
# IMPLAUSIBLE while never reasoning from the actual (correct) 5 W/kg. The
|
||||
# reasoning-text half of the gap is still open: estimate_physics() doesn't
|
||||
# return a per-metric rationale, so pass 4 still can't see WHY pass 2 landed
|
||||
# on a number, only what the number is. Upgrade path if the raw value alone
|
||||
# isn't enough in practice: have estimate_physics() also return a short
|
||||
# per-metric reason, persist it alongside raw_value (new nullable column),
|
||||
# and feed it into this prompt. Deferred because it needs a schema/interface
|
||||
# change across LLMProvider + both providers + pipeline + scorer +
|
||||
# repository, and more generated tokens per combo.
|
||||
#
|
||||
# If we plan to LLM-review every p2 pass then maybe p2 and p4 should be combined.
|
||||
#
|
||||
@@ -84,10 +120,22 @@ is NOT the question.
|
||||
{description}
|
||||
|
||||
## Metric Scores
|
||||
All scores below are normalized to 0-1, where HIGHER IS ALWAYS BETTER for
|
||||
every metric listed, regardless of what the metric measures (this already
|
||||
accounts for things like "lower cost is better" — you don't need to invert
|
||||
anything). A score of 1.0 means excellent, not "pegged" or "maxed out badly."
|
||||
Each metric below is given as its raw estimated physical value (in the unit
|
||||
shown) AND a normalized score from 0-1, where HIGHER IS ALWAYS BETTER for
|
||||
every metric listed regardless of what it measures (this already accounts
|
||||
for things like "lower cost is better" — you don't need to invert anything).
|
||||
A score of 1.0 means excellent, not "pegged" or "maxed out badly."
|
||||
|
||||
Reason from the RAW value first — it's the actual physics. The normalized
|
||||
score is a summary, not a fact on its own: a real, correct estimate can
|
||||
still normalize to a low-looking number simply because the domain's scale
|
||||
was built for a different, more demanding kind of vehicle (a cyclist's real
|
||||
~5 W/kg legitimately normalizes to ~0.16 next to a car engine's 2000 W/kg
|
||||
ceiling — that low score doesn't mean the estimate is bad or the concept is
|
||||
weak, it means human power is small next to a car engine, which everyone
|
||||
already knows). If a normalized score looks alarming, check whether the raw
|
||||
value is actually reasonable for what this component fundamentally is
|
||||
before treating the score as evidence of a problem.
|
||||
|
||||
{scores}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ from physcom.llm.prompts import (
|
||||
PHYSICS_ESTIMATION_PROMPT,
|
||||
PLAUSIBILITY_REVIEW_PROMPT,
|
||||
format_metrics_for_prompt,
|
||||
format_scores_for_prompt,
|
||||
)
|
||||
from physcom.models.domain import MetricBound
|
||||
|
||||
@@ -46,9 +47,13 @@ class GeminiLLMProvider(LLMProvider):
|
||||
return parse_metric_json(response.text, metrics)
|
||||
|
||||
def review_plausibility(
|
||||
self, combination_description: str, scores: dict[str, float]
|
||||
self,
|
||||
combination_description: str,
|
||||
raw_metrics: dict[str, float],
|
||||
normalized_scores: dict[str, float],
|
||||
metrics: list[MetricBound],
|
||||
) -> tuple[str, bool]:
|
||||
scores_str = "\n".join(f"- {k}: {v:.3f}" for k, v in scores.items())
|
||||
scores_str = format_scores_for_prompt(raw_metrics, normalized_scores, metrics)
|
||||
prompt = PLAUSIBILITY_REVIEW_PROMPT.format(
|
||||
description=combination_description,
|
||||
scores=scores_str,
|
||||
|
||||
@@ -21,9 +21,13 @@ class MockLLMProvider(LLMProvider):
|
||||
return result
|
||||
|
||||
def review_plausibility(
|
||||
self, combination_description: str, scores: dict[str, float]
|
||||
self,
|
||||
combination_description: str,
|
||||
raw_metrics: dict[str, float],
|
||||
normalized_scores: dict[str, float],
|
||||
metrics: list[MetricBound],
|
||||
) -> tuple[str, bool]:
|
||||
avg = sum(scores.values()) / max(len(scores), 1)
|
||||
avg = sum(normalized_scores.values()) / max(len(normalized_scores), 1)
|
||||
if avg > 0.5:
|
||||
return ("This concept appears plausible and worth further investigation.", True)
|
||||
return ("This concept has significant feasibility challenges.", False)
|
||||
|
||||
@@ -12,6 +12,7 @@ from physcom.llm.prompts import (
|
||||
PHYSICS_ESTIMATION_PROMPT,
|
||||
PLAUSIBILITY_REVIEW_PROMPT,
|
||||
format_metrics_for_prompt,
|
||||
format_scores_for_prompt,
|
||||
)
|
||||
from physcom.models.domain import MetricBound
|
||||
|
||||
@@ -34,9 +35,13 @@ class OllamaLLMProvider(LLMProvider):
|
||||
return parse_metric_json(text, metrics)
|
||||
|
||||
def review_plausibility(
|
||||
self, combination_description: str, scores: dict[str, float]
|
||||
self,
|
||||
combination_description: str,
|
||||
raw_metrics: dict[str, float],
|
||||
normalized_scores: dict[str, float],
|
||||
metrics: list[MetricBound],
|
||||
) -> tuple[str, bool]:
|
||||
scores_str = "\n".join(f"- {k}: {v:.3f}" for k, v in scores.items())
|
||||
scores_str = format_scores_for_prompt(raw_metrics, normalized_scores, metrics)
|
||||
prompt = PLAUSIBILITY_REVIEW_PROMPT.format(
|
||||
description=combination_description,
|
||||
scores=scores_str,
|
||||
|
||||
Reference in New Issue
Block a user