diff --git a/skills/configure/SKILL.md b/skills/configure/SKILL.md index 780410b9..0634629d 100644 --- a/skills/configure/SKILL.md +++ b/skills/configure/SKILL.md @@ -177,8 +177,8 @@ user runs the dry-run and returns its output before final approval. | Installation and authentication | `../fireworks-training/references/getting-started.md` | | Method and data selection | `../fireworks-training/references/choose-method.md` | | Preference data and evaluators | `../fireworks-training/references/preference-data-and-evaluators.md` | -| Managed RFT | `../fireworks-training/references/managed-rft-operations.md` | -| RFT tracing | `../fireworks-training/references/rft-agent-tracing.md` | +| Managed RFT (deprecated; existing jobs only) | `../fireworks-training/references/managed-rft-operations.md` | +| RFT tracing (deprecated; existing jobs only) | `../fireworks-training/references/rft-agent-tracing.md` | | Training API | `../fireworks-training/references/training-api.md` | | Training API losses | `../fireworks-training/references/training-api-losses.md` | | Secure training | `../fireworks-training/references/secure-training-operations.md` | diff --git a/skills/fireworks-training/references/choose-method.md b/skills/fireworks-training/references/choose-method.md index e0c41cfa..1f5d366b 100644 --- a/skills/fireworks-training/references/choose-method.md +++ b/skills/fireworks-training/references/choose-method.md @@ -44,7 +44,7 @@ JSONL, one object per line; OpenAI-style `messages`. **Min 3, max 3M** (aim for {"messages":[{"role":"system","content":"You are a helpful assistant."},{"role":"user","content":"Capital of France?"},{"role":"assistant","content":"Paris."}]} ``` -Docs: https://docs.fireworks.ai/fine-tuning/fine-tuning-models.md. For managed RFT weighting and launch controls, read `managed-rft-operations.md` and the installed CLI help. +Docs: https://docs.fireworks.ai/fine-tuning/fine-tuning-models.md. ## DPO format @@ -57,9 +57,11 @@ Preference pairs, **one-turn only** (preferred/non-preferred must be the last as If the user has prompts but no preference pairs, do not reject the task or silently invent labels. Use `references/preference-data-and-evaluators.md` to plan, cost, generate, review, and preserve pair provenance before upload. -## RFT — reinforcement fine-tuning +## RL — reinforcement learning -Provide three things (not necessarily labeled outputs): a **dataset** of prompts; an **evaluator or inline reward** that scores an output 0.0→1.0; and the **agent** being trained. Managed RFT uses a registered evaluator. Training API RFT uses reward code and may read `ground_truth` or any other field declared by that reward. Start with **200–500 diverse prompts**. Docs: https://docs.fireworks.ai/fine-tuning/reinforcement-fine-tuning-models.md. Evaluator authoring: `preference-data-and-evaluators.md`. +Provide three things (not necessarily labeled outputs): a **dataset** of prompts; a **reward** that scores an output 0.0→1.0; and the **agent** being trained. Reward code may read `ground_truth` or any other field it declares. Start with **200–500 diverse prompts**. Docs: https://docs.fireworks.ai/fine-tuning/training-api/cookbook/rl.md. Rollout and scheduling detail: `rl-async.md`; multi-turn agents: `rl-agentic.md`. + +Managed RFT (`firectl rftj`, registered evaluators) is deprecated and accepts no new jobs. Route new work to the Training API; use `managed-rft-operations.md` only to monitor or recover a job that already exists. ## Classification (a common SFT task) @@ -119,8 +121,7 @@ Catch format errors locally before `firectl dataset create`. A malformed row oth ```python import json, sys -method = "sft" # "sft" | "dpo" | "managed-rft" | "sdk-rft" -managed_evaluator_required_fields = [] # from the reviewed evaluator contract +method = "sft" # "sft" | "dpo" | "sdk-rft" sdk_reward_required_fields = [] # e.g. ["ground_truth"] allowed_roles = {"system", "user", "assistant", "tool"} @@ -175,10 +176,6 @@ for i, line in enumerate(open(sys.argv[1]), 1): assert len(dpo_turns) == 1 and dpo_turns[0]["role"] == "user", f"line {i}: DPO input must contain exactly one user turn" validate_preference_output(o.get("preferred_output"), i, "preferred_output") validate_preference_output(o.get("non_preferred_output"), i, "non_preferred_output") - elif method == "managed-rft": - validate_messages(o.get("messages"), i, final_assistant=False) - for field in managed_evaluator_required_fields: - assert field in o, f"line {i}: missing {field!r} required by evaluator" elif method == "sdk-rft": validate_messages(o.get("messages"), i, final_assistant=False) for field in sdk_reward_required_fields: @@ -195,7 +192,7 @@ first = json.loads(next(l for l in open(sys.argv[1]) if l.strip())) detected = "dpo" if ("preferred_output" in first or "chosen" in first) else "sft/rft" if method == "dpo" and detected != "dpo": warnings.append("requested method=dpo but rows look SFT/RFT-shaped (no preferred_output/chosen) -> wrong method or wrong file") -if method in ("sft", "managed-rft", "sdk-rft") and detected == "dpo": +if method in ("sft", "sdk-rft") and detected == "dpo": warnings.append(f"requested method={method} but rows look DPO-shaped (preferred_output/chosen present) -> wrong method") print(f"OK: {n} valid {method} rows") diff --git a/skills/fireworks-training/references/deploy-and-troubleshoot.md b/skills/fireworks-training/references/deploy-and-troubleshoot.md index ad70c2ed..d87bbb6e 100644 --- a/skills/fireworks-training/references/deploy-and-troubleshoot.md +++ b/skills/fireworks-training/references/deploy-and-troubleshoot.md @@ -15,7 +15,7 @@ A fine-tuned LoRA **cannot run on serverless** — it needs an **on-demand (dedi | Perf | Matches base | Slightly higher TTFT; lower max throughput | | Best for | Single model in prod | Experiments / many variants | -**One adapter → live merge** (simplest). Always pass a deployment shape: a bare `firectl deployment create ` drops into an **interactive shape picker**, and choosing "Create without using shape" fails with `accelerator_type must be specified for non-embeddings engines`. The interactive prompt also breaks non-interactive / agent / CI use, so pass the shape explicitly and add `--wait`. Find a deployable shape first: +**One adapter → live merge** (simplest). Always pass a deployment shape: a bare `firectl deployment create ` drops into an **interactive shape picker**, and choosing "Create without using shape" fails with `accelerator_type must be specified for non-embeddings engines`. The interactive prompt also breaks non-interactive / agent / CI use, so pass the shape explicitly and add `--wait`. Either find a deployable shape, or pass `--deployment-shape default` to let the server pick one: ```bash firectl deployment-shape-version match --model "accounts//models/" ``` diff --git a/skills/fireworks-training/references/managed-rft-operations.md b/skills/fireworks-training/references/managed-rft-operations.md index 36e583b9..36b70477 100644 --- a/skills/fireworks-training/references/managed-rft-operations.md +++ b/skills/fireworks-training/references/managed-rft-operations.md @@ -1,8 +1,14 @@ -# Managed RFT: launch, monitor, and validate +# Managed RFT: launch, monitor, and validate (deprecated) -*Source of truth: live [RFT overview](https://docs.fireworks.ai/fine-tuning/reinforcement-fine-tuning-models.md), [Models matrix](https://docs.fireworks.ai/fine-tuning/models.md), [RFT parameters](https://docs.fireworks.ai/fine-tuning/rft-parameters-reference.md), and [Eval Protocol](https://evalprotocol.io/introduction). Defer flags and defaults to installed CLI `--help`.* +> **Managed RFT is deprecated. Do not route a new run here.** Reinforcement learning +> has moved to the Training API: read `references/rl-async.md` and +> [Cookbook: Reinforcement Learning](https://docs.fireworks.ai/fine-tuning/training-api/cookbook/rl.md). +> This reference is retained only for jobs that already exist — monitoring, resuming, +> and recovering them. -Use this reference for managed RFT preflight, launch, job states, monitoring, and recovery. Training API or cookbook RL belongs in `references/training-api.md` and `references/rl-async.md`. +*Source of truth: live [RFT overview](https://docs.fireworks.ai/fine-tuning/reinforcement-fine-tuning-models.md), [Models matrix](https://docs.fireworks.ai/fine-tuning/models.md), [RL parameters](https://docs.fireworks.ai/fine-tuning/rft-parameters-reference.md), and [Eval Protocol](https://evalprotocol.io/introduction). Defer flags and defaults to installed CLI `--help`.* + +Use this reference for managed RFT job states, monitoring, and recovery on existing jobs. Training API or cookbook RL belongs in `references/training-api.md` and `references/rl-async.md`. ## Preflight diff --git a/skills/fireworks-training/references/preference-data-and-evaluators.md b/skills/fireworks-training/references/preference-data-and-evaluators.md index ee740750..febb3ef2 100644 --- a/skills/fireworks-training/references/preference-data-and-evaluators.md +++ b/skills/fireworks-training/references/preference-data-and-evaluators.md @@ -11,7 +11,7 @@ Generate preference pairs and evaluators transparently in the user's workspace: | Ideal labeled answers | SFT. Do not manufacture preference pairs. | | Human or model-ranked pairs | DPO or ORPO. Normalize to the managed preference schema. | | Prompts only, plus a clear preference criterion | Generate pairs, review a sample, then run DPO or ORPO. | -| Prompts plus objective correctness | Managed RFT with a registered evaluator, or Training API RFT with an inline reward. | +| Prompts plus objective correctness | Training API RL with an inline reward. | | Open-ended quality criteria | Write and calibrate an LLM-judge rubric before training. | Never silently turn prompts into preference data. Pair generation adds inference cost and embeds the generator or judge's bias into the training set. @@ -87,9 +87,9 @@ Dependencies and network/credential requirements: Show the spec to the user and resolve ambiguity before implementing the evaluator. -### Managed RFT evaluator +### Managed RFT evaluator (deprecated) -Managed RFT uses a registered evaluator with a reviewed entry point. Use Eval Protocol's current code-first flow, read `managed-rft-operations.md`, and defer exact APIs to the live [RFT overview](https://docs.fireworks.ai/fine-tuning/reinforcement-fine-tuning-models.md). +Managed RFT is deprecated and accepts no new jobs; for new work write a reward inside a Training API rollout function instead (see `rl-async.md`). The rest of this section applies only to evaluators already registered against an existing job. Managed RFT uses a registered evaluator with a reviewed entry point. Use Eval Protocol's current code-first flow, read `managed-rft-operations.md`, and defer exact APIs to the live [RFT overview](https://docs.fireworks.ai/fine-tuning/reinforcement-fine-tuning-models.md). 1. Write the Eval Protocol reward in the workspace. 2. Add deterministic unit examples for full credit, partial credit, zero, malformed output, and edge cases. diff --git a/skills/fireworks-training/references/rft-agent-tracing.md b/skills/fireworks-training/references/rft-agent-tracing.md index 21919358..b060eded 100644 --- a/skills/fireworks-training/references/rft-agent-tracing.md +++ b/skills/fireworks-training/references/rft-agent-tracing.md @@ -1,8 +1,12 @@ -# Managed RFT remote tracing +# Managed RFT remote tracing (deprecated) + +> **Managed RFT is deprecated.** For new multi-turn agent work use `references/rl-agentic.md` +> and [Cookbook: Agentic Reinforcement Learning](https://docs.fireworks.ai/fine-tuning/training-api/cookbook/agentic-rl.md). +> This reference covers remote environments attached to jobs that already exist. *Source of truth: live [Remote Environment Setup](https://docs.fireworks.ai/fine-tuning/connect-environments.md) and [Eval Protocol](https://evalprotocol.io/introduction).* -Use this reference when implementing a managed RFT remote environment, wiring Fireworks tracing, or debugging a reward-to-rollout join. For custom Training API agent trajectories, use `references/rl-agentic.md`. +Use this reference when debugging a reward-to-rollout join on an existing managed RFT remote environment. For custom Training API agent trajectories, use `references/rl-agentic.md`. ## Why tracing matters diff --git a/skills/fireworks-training/references/sdk-shapes.md b/skills/fireworks-training/references/sdk-shapes.md index 323ec3b7..7e0c4f93 100644 --- a/skills/fireworks-training/references/sdk-shapes.md +++ b/skills/fireworks-training/references/sdk-shapes.md @@ -37,7 +37,7 @@ trainer/deployment provisioning path. ## Deployment shape -**Do not create deployments without a [shape](https://docs.fireworks.ai/faq-new/deployment-infrastructure/what-is-a-deployment-shape).** Shapeless deployments are the most common cause of failed deployment creations, and the shapeless path may be deprecated in the future. +**Do not create deployments without a [shape](https://docs.fireworks.ai/faq-new/deployment-infrastructure/what-is-a-deployment-shape).** Shapeless deployments are the most common cause of failed deployment creations, and the [shapeless path](https://docs.fireworks.ai/guides/ondemand-deployments#explicitly-creating-a-deployment-without-a-shape-advanced-users-only) will be deprecated. Find a deployable shape, or pass `default` to let the server pick one. Do not set `cfg.deployment.deployment_shape` manually. The SDK resolves it from the requested [shape](https://docs.fireworks.ai/faq-new/deployment-infrastructure/what-is-a-deployment-shape) or the selected training profile, and recipes read diff --git a/skills/fireworks-training/references/training-api.md b/skills/fireworks-training/references/training-api.md index ef0f7d0d..995b3d83 100644 --- a/skills/fireworks-training/references/training-api.md +++ b/skills/fireworks-training/references/training-api.md @@ -14,7 +14,7 @@ Before selecting a Training API path, confirm that the target account was enable ## Managed training vs Training API -Use **managed training** for standard SFT/DPO/ORPO/RFT jobs. Reach for the **Training API** when you need a custom **loss/reward**, **RL with rollouts** (inference-in-the-loop), forward-pass internals (for example MoE routing for R3), distillation, or multi-turn/agentic trajectories. +Use **managed training** for standard SFT/DPO/ORPO jobs. Reach for the **Training API** when you need a custom **loss/reward**, **RL with rollouts** (inference-in-the-loop), forward-pass internals (for example MoE routing for R3), distillation, or multi-turn/agentic trajectories. ## Training API infrastructure @@ -25,18 +25,18 @@ Use **managed training** for standard SFT/DPO/ORPO/RFT jobs. Reach for the **Tra Read the live [serverless](https://docs.fireworks.ai/fine-tuning/training-api/serverless.md) and [dedicated](https://docs.fireworks.ai/fine-tuning/training-api/dedicated.md) pages before choosing. -## Two agent-drivable ways to run RFT/RL +## How to run RL -There are two RFT paths, and they differ in **where the reward lives**. This matters a lot when a coding agent is driving: +RL runs on the Training API. Fork `training.recipes.rl_loop` / `async_rl_loop` and supply an +inline `reward_fn(completion, row) -> float`. It may read `ground_truth`, another declared +reference field, tool outcomes, environment state, or a judge result. There is no evaluator +resource — same shape as Tinker's reward-in-the-loop — and the SDK provisions the trainer plus +rollout deployment. This is fully agent-drivable. -| Path | Reward | Agent-drivable? | -|---|---|---| -| **Managed RFT** — `firectl reinforcement-fine-tuning-job create --evaluator ` | A **registered evaluator resource** (server-side, built in an e2b sandbox, eval v3) | **Yes once the evaluator exists.** Register via **eval-protocol** (`pytest` auto-registers) or the **UI**. Evaluator authoring may require an admin role; a scoped key can still launch with an evaluator it can access. `firectl evaluator create` (V1) is **deprecated**. | -| **Training-API RL** — fork `training.recipes.rl_loop` / `async_rl_loop` | An **inline `reward_fn(completion, row) -> float`** in the forked recipe. It may read `ground_truth`, another declared reference field, tool outcomes, environment state, or a judge result. | **Yes.** No evaluator resource. Same shape as Tinker's reward-in-the-loop. The SDK provisions the trainer + rollout deployment. | - -**Prefer the managed path for standard RFT** (same as the managed UI): `firectl reinforcement-fine-tuning-job create --dataset --evaluator ` — it resolves the training shape for you and is proven live (qwen3-4b, 2026-07-15). Reuse an existing evaluator or author one via eval-protocol. **Use the inline-reward recipe (below) for users with Training API access** who need a custom loop/reward, rollouts, or agentic trajectories. Both paths are agent-drivable; they differ in reward location, access, billing, and capability. +**Managed RFT (`firectl reinforcement-fine-tuning-job create --evaluator `) is deprecated** +and accepts no new jobs. The evaluator material below applies only to jobs that already exist. -### Managed RFT: authoring the eval3 evaluator +### Managed RFT: authoring the eval3 evaluator (deprecated) `firectl reinforcement-fine-tuning-job create` needs an **eval3 evaluator with an `entry_point`** — legacy evaluators are rejected (`InvalidArgument: managed RFT requires an eval3 evaluator`), and `firectl evaluator create` is deprecated. The code-first way to make one is **eval-protocol** (no UI). Check current evaluator authorization in the live docs and handle the observed role gate below: diff --git a/skills/research/references/case-studies.md b/skills/research/references/case-studies.md index 7df6d39c..118b4aa0 100644 --- a/skills/research/references/case-studies.md +++ b/skills/research/references/case-studies.md @@ -8,7 +8,7 @@ Runnable end-to-end notebooks in `training/case-studies/`. Each README has an | `sft_prompt_router` | SFT / classification | End-to-end fine-tuning on a gradeable classification task | `prompt_router_dedicated.ipynb`, `prompt_router_serverless.ipynb` | managed SDK, serverless | | `sft_cord_receipts` | Vision SFT | One right output shape (JSON, tags, codes) from examples; invoice/OCR/form extraction | `cord_receipt_sft_sdk.ipynb` | managed SDK | | `dpo_style` | DPO | Accurate but wrong tone; easier to rank two answers than write the ideal one | `dpo_helpsteer3_sdk.ipynb` | managed SDK | -| `reasoning_rl` | GRPO / managed RFT | Objectively checkable answers; grader exists but no gold worked solutions | `rft_grpo_math.ipynb` | managed RFT | +| `reasoning_rl` | GRPO | Objectively checkable answers; grader exists but no gold worked solutions | `rft_grpo_math.ipynb` (legacy managed RFT) | New runs: Training API `training/examples/rl/deepmath/` | | `embedding_support_search` | Contrastive embedding | RAG returns adjacent but wrong article; policy structure not in base model | `airbnb_policy_embedding.ipynb` | Training API `embedding_loop` | | `agentic_rl_text2sql` | GRPO / serverless RL | Tool-calling agent (SQL, APIs); multi-turn rollouts with verifiable rewards | `sql_agent_rl_loop.ipynb` | serverless Training API | | `multilora_fleet` | LoRA SFT / multi-LoRA serving | Many tenants or locales sharing one base model; per-tenant adapters served from a single deployment | `multilora_fleet.ipynb` | managed SDK | @@ -22,7 +22,7 @@ Cookbook table: [`training/README.md`](https://github.com/fw-ai/cookbook/blob/ma | `sft_prompt_router` | SFT | managed SDK | | `sft_cord_receipts` | SFT | managed SDK | | `dpo_style` | DPO | managed SDK | -| `reasoning_rl` | RFT (GRPO) | managed RFT | +| `reasoning_rl` | RL (GRPO) | Training API `training/examples/rl/deepmath/`; `rft_grpo_math.ipynb` is legacy managed RFT | | `embedding_support_search` | embedding fine-tune | Training API dedicated | | `agentic_rl_text2sql` | RL (GRPO) | serverless Training API | | `multilora_fleet` | SFT (LoRA) | managed SDK | diff --git a/training/README.md b/training/README.md index da309493..c03e99bb 100644 --- a/training/README.md +++ b/training/README.md @@ -150,7 +150,7 @@ When `training_shape_id` is not set, the SDK selects validated runtime defaults. Explicit `training_shape_id`, `reference_training_shape_id`, and deployment-shape overrides still take precedence. -**Do not create deployments without a [shape](https://docs.fireworks.ai/faq-new/deployment-infrastructure/what-is-a-deployment-shape).** Shapeless deployments are the most common cause of failed deployment creations, and the shapeless path may be deprecated in the future. The cookbook resolves the deployment shape from the training shape profile automatically; find deployable shapes for a model with `firectl deployment-shape-version match --model `. Fields such as replica count override a shape when set; the shape owns accelerator selection. +**Do not create deployments without a [shape](https://docs.fireworks.ai/faq-new/deployment-infrastructure/what-is-a-deployment-shape).** Shapeless deployments are the most common cause of failed deployment creations, and the shapeless path will be deprecated. Find deployable shapes for a model with `firectl deployment-shape-version match --model ` (or pass `default` to let the server pick one). The cookbook resolves the deployment shape from the training shape profile automatically. Fields such as replica count override a shape when set; the shape owns accelerator selection. To launch trainers with replicated HSDP, set the run-level replica count on `TrainerConfig`; it is not part of the validated training shape: diff --git a/training/examples/rl/harbor/recipes/textworld/README.md b/training/examples/rl/harbor/recipes/textworld/README.md index 05d64025..fe94e930 100644 --- a/training/examples/rl/harbor/recipes/textworld/README.md +++ b/training/examples/rl/harbor/recipes/textworld/README.md @@ -147,6 +147,24 @@ recipe uses the shared AdamW optimizer rather than the paper's SGD setup; it isolates the score-centering correction while preserving the TextWorld comparison contract. +## Async off-policy sweeps + +Omit `--full-sync` and set the maximum rollout lead explicitly to study +staleness: + +```bash +uv run python -m training.examples.rl.harbor.recipes.textworld.train \ + ...same arguments as above... \ + --policy-loss dppo \ + --max-head-offpolicy-versions 8 \ + --epochs 2 +``` + +`--max-head-offpolicy-versions` controls how many optimizer versions rollout +production may run ahead of training. `--epochs 2` reuses the same frozen, +shuffled task order for a second pass. Use a fresh run directory for every +setting, and keep the seed and task manifest fixed when comparing values. + All policy losses use AdamW with `beta1=0.9`, `beta2=0.95`, `eps=1e-12`, and weight decay `0.01`. Gradient clipping is disabled by default (`--grad-clip-norm 0`). The recipe requests basic trainer-side gradient telemetry by default and logs diff --git a/training/examples/rl/harbor/recipes/textworld/train.py b/training/examples/rl/harbor/recipes/textworld/train.py index fa98304c..7d7b705f 100644 --- a/training/examples/rl/harbor/recipes/textworld/train.py +++ b/training/examples/rl/harbor/recipes/textworld/train.py @@ -203,6 +203,21 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: "(max_head_offpolicy_versions=0)" ), ) + parser.add_argument( + "--max-head-offpolicy-versions", + type=int, + default=MAX_HEAD_OFFPOLICY_VERSIONS, + help=( + "Maximum number of optimizer versions that rollout production may " + "run ahead; ignored when --full-sync is set" + ), + ) + parser.add_argument( + "--epochs", + type=int, + default=1, + help="Number of passes over the frozen training task order", + ) parser.add_argument("--max-concurrent-trials", type=int, default=128) parser.add_argument("--template-concurrency", type=int, default=8) parser.add_argument("--e2b-request-timeout", type=float, default=900.0) @@ -247,11 +262,14 @@ def _validate_args(args: argparse.Namespace) -> None: "completions_per_prompt", "prompt_groups_per_step", "pipeline_chunks_per_step", + "epochs", ): if getattr(args, name) < 1: raise ValueError(f"--{name.replace('_', '-')} must be positive") if args.e2b_request_timeout <= 0: raise ValueError("--e2b-request-timeout must be positive") + if args.max_head_offpolicy_versions < 0: + raise ValueError("--max-head-offpolicy-versions must be non-negative") if args.max_rows is not None and args.max_rows < 1: raise ValueError("--max-rows must be positive") if args.learning_rate < 0: @@ -303,13 +321,13 @@ def _build_config( max_completion_tokens=args.max_completion_tokens, max_seq_len=args.max_seq_len, temperature=args.temperature, - epochs=1, + epochs=args.epochs, max_rows=row_count, shuffle=False, seed=0, lora_rank=0, max_head_offpolicy_versions=( - 0 if args.full_sync else MAX_HEAD_OFFPOLICY_VERSIONS + 0 if args.full_sync else args.max_head_offpolicy_versions ), max_concurrency_rollout_sample=None, router_replay=True, @@ -564,9 +582,8 @@ def run() -> None: "evaluation_completions_per_prompt": args.eval_completions_per_prompt, "prompt_groups_per_step": args.prompt_groups_per_step, "pipeline_chunks_per_step": args.pipeline_chunks_per_step, - "max_head_offpolicy_versions": ( - 0 if args.full_sync else MAX_HEAD_OFFPOLICY_VERSIONS - ), + "epochs": config.epochs, + "max_head_offpolicy_versions": config.max_head_offpolicy_versions, "hot_load_transition_type": "SYNC" if args.full_sync else None, "training_shape_id": args.training_shape_id, "trainer_job_id": args.trainer_job_id, diff --git a/training/examples/rl/harbor/tito/trial.py b/training/examples/rl/harbor/tito/trial.py index aa51a003..03763c39 100644 --- a/training/examples/rl/harbor/tito/trial.py +++ b/training/examples/rl/harbor/tito/trial.py @@ -499,6 +499,7 @@ def _build_trial_config( agent_provider: str = "fireworks-rl", tool_timeout_seconds: int = DEFAULT_HARNESS_TOOL_TIMEOUT_SECONDS, tool_profile: str = "coding", + collect_tito_artifacts: bool = True, ) -> Any: """Merge a native TrialConfig template with Fireworks-owned runtime fields.""" @@ -577,31 +578,36 @@ def _build_trial_config( agent["kwargs"] = agent_kwargs artifacts = list(document.get("artifacts") or ()) - for source in ( - SIDECAR_ARTIFACT_PATH, - SIDECAR_ARTIFACT_MANIFEST_PATH, - SIDECAR_COMPLETE_PATH, - ): - artifacts.append( - { - "source": source, - "destination": str(_COMPACT_ARTIFACT_DESTINATION / Path(source).name), - } - ) - for source in (SIDECAR_STDOUT_PATH, SIDECAR_STDERR_PATH): - artifacts.append( - { - "source": source, - "destination": str(_LOG_ARTIFACT_DESTINATION / Path(source).name), - } - ) - if bool(sidecar_spec.get("debug_enabled")): - artifacts.append( - { - "source": SIDECAR_DEBUG_ROOT, - "destination": str(_DEBUG_ARTIFACT_DESTINATION), - } - ) + if collect_tito_artifacts: + for source in ( + SIDECAR_ARTIFACT_PATH, + SIDECAR_ARTIFACT_MANIFEST_PATH, + SIDECAR_COMPLETE_PATH, + ): + artifacts.append( + { + "source": source, + "destination": str( + _COMPACT_ARTIFACT_DESTINATION / Path(source).name + ), + } + ) + for source in (SIDECAR_STDOUT_PATH, SIDECAR_STDERR_PATH): + artifacts.append( + { + "source": source, + "destination": str( + _LOG_ARTIFACT_DESTINATION / Path(source).name + ), + } + ) + if bool(sidecar_spec.get("debug_enabled")): + artifacts.append( + { + "source": SIDECAR_DEBUG_ROOT, + "destination": str(_DEBUG_ARTIFACT_DESTINATION), + } + ) document["artifacts"] = artifacts document.update( @@ -823,6 +829,7 @@ async def run_harbor_trial( agent_version=agent_version, tool_timeout_seconds=tool_timeout_seconds, tool_profile=tool_profile, + collect_tito_artifacts=require_trajectory_artifact, ) result = None trial_path = trial_root / config.trial_name diff --git a/training/renderer/_kimi_k25_split.py b/training/renderer/_kimi_k25_split.py index 35f7948a..a59193c4 100644 --- a/training/renderer/_kimi_k25_split.py +++ b/training/renderer/_kimi_k25_split.py @@ -28,10 +28,15 @@ from training.renderer.tokenizer import Tokenizer from training.renderer._disaggregate_mixin import DisaggregateMultiTurnMixin +from training.renderer.kimi_k26 import _KimiMediaPadImagePlaceholderMixin from training.renderer.message_weights import untrained_synthesized_context -class KimiK25SplitRenderer(DisaggregateMultiTurnMixin, _TinkerKimiK25Renderer): +class KimiK25SplitRenderer( + _KimiMediaPadImagePlaceholderMixin, + DisaggregateMultiTurnMixin, + _TinkerKimiK25Renderer, +): """Upstream K2.5 rendering with per-turn, weight-aware loss placement.""" def _ensure_system_message(self, messages: list[Message]) -> list[Message]: diff --git a/training/renderer/glm5.py b/training/renderer/glm5.py index 4702eb9b..a3f1a384 100644 --- a/training/renderer/glm5.py +++ b/training/renderer/glm5.py @@ -554,16 +554,39 @@ class GLM5Renderer(DisaggregateMultiTurnMixin, Renderer): _historical_stripped_think_block = "" _preserve_has_extension_property = False + # Map the API-style effort vocabulary onto the tiers this model family's + # template can actually render, the same way the serving conversation style + # does. GLM-5.1's template has no reasoning-effort line at all, so the base + # renderer accepts no effort; subclasses whose templates render one declare + # their own vocabulary. + _EFFORT_TIERS: Mapping[str, str] = {} + def __init__( self, tokenizer: Tokenizer, *, clear_thinking: bool = True, honor_source_reasoning_fields: bool = False, + reasoning_effort: str | None = None, ) -> None: super().__init__(tokenizer) self._clear_thinking = clear_thinking self._honor_source_reasoning_fields = honor_source_reasoning_fields + if reasoning_effort is not None: + if not self._EFFORT_TIERS: + raise ValueError( + f"{type(self).__name__} renders no reasoning-effort system " + f"line; reasoning_effort={reasoning_effort!r} is unsupported" + ) + tier = self._EFFORT_TIERS.get(str(reasoning_effort).strip().lower()) + if tier is None: + raise ValueError( + f"unknown reasoning_effort {reasoning_effort!r}; " + f"expected one of {sorted(self._EFFORT_TIERS)}" + ) + # Instance attribute shadows the class-level default template line + # (``_initial_prompt_tokens`` reads it at render time). + self._initial_prompt_text = f"<|system|>Reasoning Effort: {tier}" @property def has_extension_property(self) -> bool: @@ -1104,6 +1127,17 @@ class GLMMoeDsaRenderer(GLM5Renderer): _historical_stripped_think_block = "" _preserve_has_extension_property = True + # GLM-5.2's template renders only High (for template value ``high``) and + # Max (for everything else), so ``low`` folds into High exactly as serving + # does. Writing ``Low`` here would train on a prefix the served prompt for + # a low-effort request never contains. + _EFFORT_TIERS: Mapping[str, str] = { + "low": "High", + "medium": "High", + "high": "High", + "max": "Max", + } + class GLM53Renderer(GLMMoeDsaRenderer): """Renderer for the pinned ``zai-org/GLM-5.3`` chat-template contract. @@ -1118,17 +1152,26 @@ class GLM53Renderer(GLMMoeDsaRenderer): # independently for each message. supports_per_message_rendering = False + # GLM-5.3's template adds the Low tier GLM-5.2 lacks; the rest of the + # vocabulary is unchanged. + _EFFORT_TIERS: Mapping[str, str] = { + **GLMMoeDsaRenderer._EFFORT_TIERS, + "low": "Low", + } + def __init__( self, tokenizer: Tokenizer, *, clear_thinking: bool = False, honor_source_reasoning_fields: bool = True, + reasoning_effort: str | None = None, ) -> None: super().__init__( tokenizer, clear_thinking=clear_thinking, honor_source_reasoning_fields=honor_source_reasoning_fields, + reasoning_effort=reasoning_effort, ) def build_supervised_example( @@ -1182,11 +1225,13 @@ def __init__( image_processor: Any | None = None, clear_thinking: bool = False, honor_source_reasoning_fields: bool = True, + reasoning_effort: str | None = None, ) -> None: super().__init__( tokenizer, clear_thinking=clear_thinking, honor_source_reasoning_fields=honor_source_reasoning_fields, + reasoning_effort=reasoning_effort, ) self.image_processor = image_processor image_special_tokens = { diff --git a/training/renderer/kimi_k26.py b/training/renderer/kimi_k26.py index 046c3169..642d6a24 100644 --- a/training/renderer/kimi_k26.py +++ b/training/renderer/kimi_k26.py @@ -86,7 +86,55 @@ def render_message( return super().render_message(message, ctx) # type: ignore[misc] +# Distinguishes "not resolved yet" from a resolved ``None``. +_UNRESOLVED = object() + + +class _KimiMediaPadImagePlaceholderMixin: + """Resolve the image-placeholder token id for token-in vision completions. + + The K2.5/K2.6 renderer lineage is text-only, so nothing up the MRO + supplies this. Rollout multimodal rendering needs it to encode image + chunks for token-in completions, and without it a vision-capable + checkpoint samples zero multimodal prompt groups and RL fails the "no + trained multimodal prompt group" gate. Resolve it the way + ``KimiK3VisionRenderer`` does -- from the tokenizer's ``<|media_pad|>`` + special token -- and stay ``None`` for tokenizers that lack it rather + than returning an ``unk`` id that would silently render as text. + """ + + @property + def image_placeholder_token_id(self) -> int | None: + """Token id standing in for one image chunk, or ``None`` when text-only.""" + cached = getattr(self, "_image_placeholder_token_id_cache", _UNRESOLVED) + if cached is not _UNRESOLVED: + return cached + + # Imported lazily: ``renderer/__init__`` loads this module before + # ``kimi_k3``, so a module-level import would close a cycle. + from training.renderer.kimi_k3 import MEDIA_PAD_TOKEN + + resolved: int | None = None + convert = getattr(self.tokenizer, "convert_tokens_to_ids", None) + if callable(convert): + try: + candidate = convert(MEDIA_PAD_TOKEN) + except (KeyError, ValueError): + candidate = None + if isinstance(candidate, int) and not isinstance(candidate, bool): + # A tokenizer without the token maps it to ``unk`` rather than + # failing, and encoding that id would render the image chunk as + # ordinary text instead of a placeholder. + unk_id = getattr(self.tokenizer, "unk_token_id", None) + if not (isinstance(unk_id, int) and candidate == unk_id): + resolved = candidate + + self._image_placeholder_token_id_cache = resolved + return resolved + + class KimiK25InterleavedRenderer( + _KimiMediaPadImagePlaceholderMixin, _KimiReasoningFieldPrecedenceMixin, DisaggregateMultiTurnMixin, _NoImplicitSystemMessageMixin, @@ -108,6 +156,7 @@ class KimiK26InterleavedRenderer(KimiK25InterleavedRenderer): class KimiK26PreserveThinkingRenderer( + _KimiMediaPadImagePlaceholderMixin, _KimiReasoningFieldPrecedenceMixin, DisaggregateMultiTurnMixin, _NoImplicitSystemMessageMixin, diff --git a/training/renderer/kimi_k27_code.py b/training/renderer/kimi_k27_code.py index 4f421994..28c84cf8 100644 --- a/training/renderer/kimi_k27_code.py +++ b/training/renderer/kimi_k27_code.py @@ -34,6 +34,7 @@ from training.renderer._disaggregate_mixin import DisaggregateMultiTurnMixin from training.renderer.kimi_k26 import ( KimiK26PreserveThinkingRenderer as _CookbookKimiK26PreserveThinkingRenderer, + _KimiMediaPadImagePlaceholderMixin, ) @@ -49,10 +50,6 @@ ] -# Distinguishes "not resolved yet" from a resolved ``None``. -_UNRESOLVED = object() - - def _tokenizer_tools_branch_uses_typescript(tokenizer: Any) -> bool: apply_chat_template = getattr(tokenizer, "apply_chat_template", None) if apply_chat_template is None: @@ -71,46 +68,13 @@ def _tokenizer_tools_branch_uses_typescript(tokenizer: Any) -> bool: return isinstance(rendered, str) and "namespace functions" in rendered -class _KimiK27CodeMixin: - """K2.7-specific system/tool declaration behavior.""" - - @property - def image_placeholder_token_id(self) -> int | None: - """Token id standing in for one image chunk, or ``None`` when text-only. - - K2.7 inherits a text-only renderer lineage from K2.6, so nothing up the - MRO resolves this. Rollout multimodal rendering needs it to encode image - chunks for token-in completions, and without it a vision-capable K2.7 - checkpoint samples zero multimodal prompt groups. Resolve it the way - ``KimiK3VisionRenderer`` does -- from the tokenizer's ``<|media_pad|>`` - special token -- and stay ``None`` for tokenizers that lack it rather - than returning an ``unk`` id that would silently render as text. - """ - cached = getattr(self, "_image_placeholder_token_id_cache", _UNRESOLVED) - if cached is not _UNRESOLVED: - return cached - - # Imported lazily: ``renderer/__init__`` loads this module before - # ``kimi_k3``, so a module-level import would close a cycle. - from training.renderer.kimi_k3 import MEDIA_PAD_TOKEN - - resolved: int | None = None - convert = getattr(self.tokenizer, "convert_tokens_to_ids", None) - if callable(convert): - try: - candidate = convert(MEDIA_PAD_TOKEN) - except (KeyError, ValueError): - candidate = None - if isinstance(candidate, int) and not isinstance(candidate, bool): - # A tokenizer without the token maps it to ``unk`` rather than - # failing, and encoding that id would render the image chunk as - # ordinary text instead of a placeholder. - unk_id = getattr(self.tokenizer, "unk_token_id", None) - if not (isinstance(unk_id, int) and candidate == unk_id): - resolved = candidate - - self._image_placeholder_token_id_cache = resolved - return resolved +class _KimiK27CodeMixin(_KimiMediaPadImagePlaceholderMixin): + """K2.7-specific system/tool declaration behavior. + + ``image_placeholder_token_id`` comes from the shared media-pad mixin: K2.7 + inherits a text-only renderer lineage from K2.6, so nothing up the MRO + would resolve it otherwise. + """ def _ensure_system_message(self, messages: list[Message]) -> list[Message]: return list(messages) diff --git a/training/tests/unit/test_glm53_renderer.py b/training/tests/unit/test_glm53_renderer.py index 572b7c2c..9b578548 100644 --- a/training/tests/unit/test_glm53_renderer.py +++ b/training/tests/unit/test_glm53_renderer.py @@ -13,7 +13,7 @@ import training.renderer.glm5 # noqa: F401 - registers glm53 from training.renderer import RendererError, get_renderer -from training.renderer.glm5 import Glm53FlashImageTokenCounter +from training.renderer.glm5 import GLM53Renderer, Glm53FlashImageTokenCounter from training.utils.rl.rollout.renderer import ( build_multimodal_completions_prompt_token_ids, ) @@ -676,3 +676,35 @@ def test_flash_invalid_tool_result_block_matches_current_hf( ) rendered = flash_tokenizer.decode(ours) assert rendered.index("first") < rendered.index("second") + + +def test_reasoning_effort_overrides_initial_prompt_line(tokenizer): + messages = [{"role": "user", "content": "hi"}] + + default = GLM53Renderer(tokenizer) + assert default._initial_prompt_text == "<|system|>Reasoning Effort: Max" + assert "Reasoning Effort: Max" in tokenizer.decode( + default.build_generation_prompt(list(messages)).to_ints() + ) + + low = GLM53Renderer(tokenizer, reasoning_effort="low") + rendered = tokenizer.decode(low.build_generation_prompt(list(messages)).to_ints()) + assert "Reasoning Effort: Low" in rendered + assert "Max" not in rendered.split("Reasoning Effort:")[1].split("\n")[0] + + # medium/high fold into the template's High tier, mirroring the serving + # conversation-style mapping. + high = GLM53Renderer(tokenizer, reasoning_effort="medium") + assert "Reasoning Effort: High" in tokenizer.decode( + high.build_generation_prompt(list(messages)).to_ints() + ) + + max_effort = GLM53Renderer(tokenizer, reasoning_effort="max") + assert max_effort.build_generation_prompt( + list(messages) + ).to_ints() == default.build_generation_prompt(list(messages)).to_ints() + + +def test_reasoning_effort_rejects_unknown_tier(tokenizer): + with pytest.raises(ValueError, match="unknown reasoning_effort"): + GLM53Renderer(tokenizer, reasoning_effort="ultra") diff --git a/training/tests/unit/test_glm5_renderer.py b/training/tests/unit/test_glm5_renderer.py index 8c02123c..fb5b35bb 100644 --- a/training/tests/unit/test_glm5_renderer.py +++ b/training/tests/unit/test_glm5_renderer.py @@ -117,6 +117,13 @@ def test_registered_glm5_preserve_thinking_does_not_claim_extension_property( assert preserve.has_extension_property is False +def test_reasoning_effort_rejected_for_glm51(tokenizer): + # GLM-5.1's template never renders a reasoning-effort system line, so any + # tier the renderer wrote would be a prefix serving cannot reproduce. + with pytest.raises(ValueError, match="renders no reasoning-effort"): + GLM5Renderer(tokenizer, reasoning_effort="low") + + def _hf_tokens(tokenizer, messages, add_generation_prompt: bool, **kwargs) -> list[int]: """Tokenize via the HF jinja template, returning a plain list of ints. diff --git a/training/tests/unit/test_glm_moe_dsa_renderer.py b/training/tests/unit/test_glm_moe_dsa_renderer.py index 10680541..7722bc36 100644 --- a/training/tests/unit/test_glm_moe_dsa_renderer.py +++ b/training/tests/unit/test_glm_moe_dsa_renderer.py @@ -11,6 +11,7 @@ from training.renderer.tito import build_sidecar_tito_renderer import training.renderer.glm5 # noqa: F401 - registers glm_moe_dsa from training.renderer import get_renderer +from training.renderer.glm5 import GLMMoeDsaRenderer _TOKENIZER = "zai-org/GLM-5.2" @@ -118,6 +119,27 @@ def test_generation_prompt_user_only_matches_hf(tokenizer, renderer): assert "Reasoning Effort: Max" in tokenizer.decode(ours) +def test_low_reasoning_effort_renders_the_served_high_tier(tokenizer): + # GLM-5.2's template has no Low tier, and serving folds a low-effort + # request onto ``high``, so the renderer must train on the High prefix. + messages = [{"role": "user", "content": "Hello"}] + low = GLMMoeDsaRenderer(tokenizer, reasoning_effort="low") + + ours = _renderer_generation_tokens(low, messages) + served = tokenizer.encode( + tokenizer.apply_chat_template( + messages, + tokenize=False, + add_generation_prompt=True, + reasoning_effort="high", + ), + add_special_tokens=False, + ) + + assert ours == served + assert "Reasoning Effort: High" in tokenizer.decode(ours) + + def test_incremental_tito_prompt_matches_full_glm52_render(tokenizer) -> None: sidecar_renderer = build_sidecar_tito_renderer( tokenizer, diff --git a/training/tests/unit/test_harbor_textworld.py b/training/tests/unit/test_harbor_textworld.py index 34a7452b..3000555e 100644 --- a/training/tests/unit/test_harbor_textworld.py +++ b/training/tests/unit/test_harbor_textworld.py @@ -199,6 +199,7 @@ def test_textworld_pi_recipe_uses_e2b_and_managed_server_grpo(tmp_path): assert config.completions_per_prompt == 8 assert config.prompt_groups_per_step == 8 + assert config.epochs == 1 assert config.max_head_offpolicy_versions == 2 assert config.anchor_logp == "rollout" assert config.server_side_grpo is True @@ -285,6 +286,41 @@ def test_textworld_full_sync_shape_and_batch_contract(tmp_path): assert config.deployment.hot_load_transition_type == "SYNC" +def test_textworld_async_offpolicy_and_epoch_controls(tmp_path) -> None: + from training.examples.rl.harbor.recipes.textworld import train as train_textworld + + args = train_textworld.parse_args( + [ + "--base-model", + "accounts/example/models/policy", + "--tokenizer-model", + "Qwen/Qwen3.8-27B", + "--renderer-name", + "qwen3_8", + "--textworld-dataset", + str(tmp_path / "dataset"), + "--run-dir", + str(tmp_path / "run"), + "--shuffle-seed", + "19", + "--max-head-offpolicy-versions", + "8", + "--epochs", + "2", + ] + ) + + config = train_textworld._build_config( + args, + run_dir=tmp_path / "run", + row_count=256, + ) + + assert config.max_head_offpolicy_versions == 8 + assert config.epochs == 2 + assert config.deployment.hot_load_transition_type is None + + @pytest.mark.parametrize( "policy_loss", ["dapo", "dro", "cispo", "dppo", "score_centering"] ) diff --git a/training/tests/unit/test_harbor_tito.py b/training/tests/unit/test_harbor_tito.py index 8808f61f..3bddc3b8 100644 --- a/training/tests/unit/test_harbor_tito.py +++ b/training/tests/unit/test_harbor_tito.py @@ -1407,6 +1407,34 @@ def test_trial_config_uses_same_sidecar_contract_for_both_backends( ] +def test_reward_only_trial_config_omits_tito_artifact_downloads(tmp_path) -> None: + config = harbor_adapter._build_trial_config( + _fake_harbor(), + template={ + "environment": {"type": "e2b"}, + "artifacts": [{"source": "/logs/verifier/reward.txt"}], + }, + task_config={"path": "/tasks/example"}, + run_id="eval", + trials_dir=tmp_path, + harbor_environment="e2b", + sidecar_bundle_path=tmp_path / "bundle", + sidecar_launch_spec=json.dumps( + { + "api_key": "secret", + "inference_base_url": "https://api.fireworks.ai", + } + ), + context_limit=4096, + output_limit=1024, + agent_import_path=OPENCODE_HARBOR_IMPORT_PATH, + agent_version=DEFAULT_OPENCODE_VERSION, + collect_tito_artifacts=False, + ) + + assert config.artifacts == [{"source": "/logs/verifier/reward.txt"}] + + def test_e2b_rejects_compose_task(tmp_path) -> None: task_path = tmp_path / "task" environment_path = task_path / "environment" diff --git a/training/tests/unit/test_kimi_history_renderers.py b/training/tests/unit/test_kimi_history_renderers.py index 5bd4bb78..9c8abf58 100644 --- a/training/tests/unit/test_kimi_history_renderers.py +++ b/training/tests/unit/test_kimi_history_renderers.py @@ -487,3 +487,73 @@ def test_preserve_generation_prompt_extends_prior_observation( second_turn_prompt = renderer.build_generation_prompt(later_prompt).to_ints() assert second_turn_prompt[: len(completed_first_turn)] == completed_first_turn + + +_MEDIA_PAD_TOKEN = "<|media_pad|>" +_MEDIA_PAD_ID = 9 +_UNK_ID = 4 + + +class _MediaPadTokenizer(_ReversibleTokenizer): + """Stub tokenizer that registers ``<|media_pad|>`` like Kimi vision models.""" + + unk_token_id = _UNK_ID + _special_to_id = { + **_ReversibleTokenizer._special_to_id, + _MEDIA_PAD_TOKEN: _MEDIA_PAD_ID, + } + _id_to_special = {value: key for key, value in _special_to_id.items()} + + def convert_tokens_to_ids(self, token: str) -> int: + return self._special_to_id.get(token, self.unk_token_id) + + +class _NoMediaPadTokenizer(_ReversibleTokenizer): + """Stub tokenizer that maps unknown tokens to ``unk`` instead of failing.""" + + unk_token_id = _UNK_ID + + def convert_tokens_to_ids(self, token: str) -> int: + return self.unk_token_id + + +@pytest.mark.parametrize( + "renderer_name", + [ + "kimi_k25", + "kimi_k25_interleaved", + "kimi_k26_interleaved", + "kimi_k26_preserve_thinking", + ], +) +def test_image_placeholder_token_id_resolves_media_pad(renderer_name: str) -> None: + """Rollout multimodal rendering needs this to encode image chunks. + + The K2.5/K2.6 renderer lineage is text-only, so without this resolution a + vision-capable checkpoint samples zero multimodal prompt groups and RL + fails the "no trained multimodal prompt group" gate. + """ + renderer = get_renderer(renderer_name, _MediaPadTokenizer()) + assert renderer.image_placeholder_token_id == _MEDIA_PAD_ID + + +@pytest.mark.parametrize( + "renderer_name", + [ + "kimi_k25", + "kimi_k25_interleaved", + "kimi_k26_interleaved", + "kimi_k26_preserve_thinking", + ], +) +def test_image_placeholder_token_id_is_none_without_media_pad( + renderer_name: str, +) -> None: + """A tokenizer lacking the token must yield None, never its ``unk`` id. + + ``convert_tokens_to_ids`` maps an unknown token to ``unk`` instead of + failing, and encoding that id would silently render an image chunk as + ordinary text. + """ + renderer = get_renderer(renderer_name, _NoMediaPadTokenizer()) + assert renderer.image_placeholder_token_id is None diff --git a/training/utils/config.py b/training/utils/config.py index 3de6b204..b4b9df74 100644 --- a/training/utils/config.py +++ b/training/utils/config.py @@ -261,7 +261,8 @@ class DeployConfig: """Deployment shape resource name. Should always be a **versioned** path (e.g. ``accounts/fw/deploymentShapes/ds-x/versions/abc123``) to pin the exact shape config. Recipes populate this from - ``profile.deployment_shape`` which returns the versioned path.""" + ``profile.deployment_shape`` which returns the versioned path. + Required: the cookbook never creates deployments without a shape.""" hot_load_bucket_type: str = "FW_HOSTED" hot_load_trainer_job: str | None = None """Trainer job name whose hot-load bucket this deployment should use. @@ -317,8 +318,8 @@ def to_deployment_config( raise ValueError( "DeployConfig.deployment_shape is required. Shapeless " "deployments are the most common cause of failed deployment " - "creations, and the shapeless path may be deprecated in the " - "future. Resolve a shape from the training shape profile " + "creations, and the shapeless path will be deprecated. " + "Resolve a shape from the training shape profile " "(``profile.deployment_shape``) or find deployable shapes " "with ``firectl deployment-shape-version match --model " "``."