Merge remote-tracking branch 'origin/main' into worktree-anti-pattern-directive-position
This commit is contained in:
@@ -371,6 +371,7 @@ async def write_block(
|
||||
block_id: str,
|
||||
instructions: str = "",
|
||||
effort_override: str | None = None,
|
||||
model_override: str | None = None,
|
||||
) -> dict:
|
||||
"""כתיבת בלוק יחיד בהחלטה.
|
||||
|
||||
@@ -383,6 +384,12 @@ async def write_block(
|
||||
THIS call only — used by the #208 model/effort calibration harness
|
||||
to A/B efforts without mutating the pinned defaults. Production
|
||||
callers leave it None and get the deterministic per-block effort.
|
||||
model_override: optional per-call generation model id (e.g.
|
||||
"claude-opus-5"). Same contract as effort_override — the #208
|
||||
harness A/Bs MODELS without mutating the pinned GENERATION_MODEL.
|
||||
Pass the BASE id only: the 1M-context escalation (#216) is applied
|
||||
on top automatically for large prompts, so an override never
|
||||
silently loses the 1M window. Production callers leave it None.
|
||||
|
||||
Returns:
|
||||
dict עם content, word_count, block_id, generation_type
|
||||
@@ -486,7 +493,12 @@ async def write_block(
|
||||
# escalate to the 1M-context build (`[1m]`) instead of failing the block —
|
||||
# block-yod legitimately carries the whole case as source-context. The 400K
|
||||
# ceiling was an artifact of the old 200K-only build, NOT a model limit.
|
||||
gen_model = GENERATION_MODEL_1M if len(prompt) > _CTX_1M_THRESHOLD_CHARS else GENERATION_MODEL
|
||||
# model_override (#208 harness) swaps the BASE id only — the 1M decision below
|
||||
# still applies, so an A/B'd model keeps the same context-window behaviour as
|
||||
# the pinned default instead of silently falling back to the 200K build.
|
||||
_base_model = model_override or GENERATION_MODEL
|
||||
_model_1m = GENERATION_MODEL_1M if _base_model == GENERATION_MODEL else f"{_base_model}[1m]"
|
||||
gen_model = _model_1m if len(prompt) > _CTX_1M_THRESHOLD_CHARS else _base_model
|
||||
|
||||
# Final guard: even the 1M build is finite (~2M Hebrew chars of input). Cap at
|
||||
# 1.5M chars (~750K tokens) to leave room for output + a safety margin under 1M.
|
||||
|
||||
Reference in New Issue
Block a user