diff --git a/.codex/environments/environment.toml b/.codex/environments/environment.toml index 4218894..e3931b9 100644 --- a/.codex/environments/environment.toml +++ b/.codex/environments/environment.toml @@ -35,6 +35,11 @@ name = "Deploy Training Data" icon = "tool" command = "scripts/deploy_training_data.sh" +[[actions]] +name = "Deploy Training Data With Mimir" +icon = "tool" +command = "ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh" + [[actions]] name = "Run Logger" icon = "run" diff --git a/.env.example b/.env.example index 52cb18d..be9e71f 100644 --- a/.env.example +++ b/.env.example @@ -9,3 +9,15 @@ LLAMA_EXTRA_ARGS= # Optional controller logging. CONTROLLER_LOG_LEVEL=info + +# Optional Mimir/Data Designer SDG settings for training-data deploys. +ENABLE_MIMIR_SDG=0 +MIMIR_SDG_HOST=100.80.52.47 +MIMIR_SDG_USER=aaron-pressey +MIMIR_SDG_KEY=~/.ssh/silma_orson_ed25519 +MIMIR_SDG_MODEL_ALIAS=nvidia-text +MIMIR_SDG_NUM_RECORDS=1000 + +# Optional provider credential for local Data Designer usage. +# Mimir stores its own copy in /home/aaron-pressey/.config/home-grown-llm-data/data-designer.env. +NVIDIA_API_KEY= diff --git a/.woodpecker.yml b/.woodpecker.yml index 9ac6919..e040aa8 100644 --- a/.woodpecker.yml +++ b/.woodpecker.yml @@ -15,7 +15,7 @@ steps: image: rust:1.95 commands: - apt-get update && apt-get install -y python3 - - scripts/deploy_training_data.sh + - ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh when: - event: push branch: main diff --git a/AGENTS.md b/AGENTS.md index eb71801..b35a865 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -22,6 +22,7 @@ - Training data dry run: `python3 scripts/generate_training_data.py --dry-run` - Local training data bridge: `python3 scripts/generate_training_data.py` - Deploy training data to Zoe: `scripts/deploy_training_data.sh` +- Deploy training data with Mimir SDG: `ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh` ## Environment @@ -33,6 +34,9 @@ Configuration is read from environment variables: - `LLAMA_CONTEXT_SIZE`: optional value passed as `--ctx-size`. - `LLAMA_EXTRA_ARGS`: optional shell-split arguments appended to `llama-server`. - `CONTROLLER_LOG_LEVEL`: tracing log level. Default: `info`. +- `ENABLE_MIMIR_SDG`: set to `1` to include Mimir/Data Designer expansion in training-data deploys. +- `MIMIR_SDG_HOST`, `MIMIR_SDG_USER`, `MIMIR_SDG_KEY`, `MIMIR_SDG_MODEL_ALIAS`, `MIMIR_SDG_NUM_RECORDS`: optional SDG deploy overrides. +- `NVIDIA_API_KEY`: optional local Data Designer provider credential; keep only in ignored `.env` files or Mimir's Data Designer env file. Use `.env` for local values and keep it out of git. Update `.env.example`, `README.md`, `manifest.llm.json`, and `llm.txt` when environment variables or commands change. @@ -58,9 +62,13 @@ Use `.env` for local values and keep it out of git. Update `.env.example`, `READ - `/Volumes/Zoe/custom-local-llm` has its own `README.md`, `AGENTS.md`, `manifest.llm.json`, and `llm.txt`. - Use `--dry-run` for validation; it must not write output data or patch manifests. - Mimir/Data Designer expansion only runs when `--sdg-host`, `--sdg-user`, and `--sdg-key` are all provided. +- `scripts/deploy_training_data.sh` enables Mimir expansion when `ENABLE_MIMIR_SDG=1`; Woodpecker uses this mode on main-branch pushes. +- Mimir expansion defaults to `MIMIR_SDG_MODEL_ALIAS=nvidia-text`; set `MIMIR_SDG_MODEL_ALIAS=openrouter-text` only if the direct NVIDIA provider needs a fallback. +- Mimir expansion sources `/home/aaron-pressey/.config/home-grown-llm-data/data-designer.env`, writes `expanded.raw.jsonl`, and normalizes usable instruction rows into `expanded.jsonl`. - Do not start or restart Mimir llama.cpp/Keiro from this bridge. - If Mimir expansion fails, the bridge keeps the local seed dataset, writes `SDG_ERROR.txt`, patches the manifest with `generated_local_remote_failed`, and exits non-zero. -- Woodpecker deploys training data on main-branch pushes with `scripts/deploy_training_data.sh`. +- Seed-only `train.jsonl` and `validation.jsonl` are routing smoke-test quality; use `expanded.jsonl` from Mimir for serious training-data enrichment. +- Woodpecker deploys training data on main-branch pushes with `ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh`. ## Dispatch Rules diff --git a/README.md b/README.md index bcb789b..043ad20 100644 --- a/README.md +++ b/README.md @@ -27,6 +27,7 @@ cargo run -- run write_note --payload '{"filename":"hello.txt","content":"hello" scripts/precommit-check.sh python3 scripts/generate_training_data.py --dry-run scripts/deploy_training_data.sh +ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh ``` `serve` starts: @@ -40,6 +41,9 @@ Optional environment variables: - `LLAMA_CONTEXT_SIZE`: appended as `--ctx-size`. - `LLAMA_EXTRA_ARGS`: shell-split and appended to `llama-server`. - `CONTROLLER_LOG_LEVEL`: tracing log level, default `info`. +- `ENABLE_MIMIR_SDG`: set to `1` for Mimir/Data Designer expansion during training-data deploy. +- `MIMIR_SDG_HOST`, `MIMIR_SDG_USER`, `MIMIR_SDG_KEY`, `MIMIR_SDG_MODEL_ALIAS`, `MIMIR_SDG_NUM_RECORDS`: SDG deploy overrides. +- `NVIDIA_API_KEY`: optional local Data Designer provider credential. Keep it in ignored `.env` only. ## Dictionary @@ -134,12 +138,14 @@ python3 scripts/generate_training_data.py \ --manifest /Volumes/Zoe/custom-local-llm/manifest.llm.json ``` -Optional Mimir/Data Designer expansion is enabled only when `--sdg-host`, `--sdg-user`, and `--sdg-key` are supplied. The bridge does not start or restart Mimir inference services. +Optional Mimir/Data Designer expansion is enabled only when `--sdg-host`, `--sdg-user`, and `--sdg-key` are supplied. The deploy helper turns this on with `ENABLE_MIMIR_SDG=1` and defaults to Mimir at `100.80.52.47`, user `aaron-pressey`, key `~/.ssh/silma_orson_ed25519`, model alias `nvidia-text`, and `1000` expanded rows. Override those with `MIMIR_SDG_HOST`, `MIMIR_SDG_USER`, `MIMIR_SDG_KEY`, `MIMIR_SDG_MODEL_ALIAS`, and `MIMIR_SDG_NUM_RECORDS`. + +Remote expansion uses SSH to run Mimir's Data Designer CLI. It sources `/home/aaron-pressey/.config/home-grown-llm-data/data-designer.env`, uses the configured model alias, copies back `expanded.raw.jsonl`, and normalizes it to `expanded.jsonl`. The bridge does not start or restart Mimir inference services. If remote expansion fails, the timestamped local seed dataset remains in place, `SDG_ERROR.txt` is written in that dataset directory, and the manifest entry is marked `generated_local_remote_failed`. -Woodpecker deploys a new training-data export to Zoe on `main` pushes by running: +Woodpecker deploys a new enriched training-data export to Zoe on `main` pushes by running: ```sh -scripts/deploy_training_data.sh +ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh ``` diff --git a/SDG_BRIDGE_PLAN.md b/SDG_BRIDGE_PLAN.md index 424fad8..0d8ee5c 100644 --- a/SDG_BRIDGE_PLAN.md +++ b/SDG_BRIDGE_PLAN.md @@ -26,9 +26,10 @@ Mimir (`100.80.52.47`) has **Data Designer 0.8.0** installed and ready: | Model configs | `/mnt/storage/data-designer/model_configs.yaml` | ✅ configured | | Config env | `/home/aaron-pressey/.config/home-grown-llm-data/data-designer.env` | ✅ present | -**Current model provider:** NVIDIA API (`nvidia-text` alias → `nvidia/nemotron-3-nano-30b-a3b`). -Local llama.cpp on port 8081 is configured but not currently running — cloud provider is used -for generation unless a local inference server is started separately. +**Current model provider:** NVIDIA API through the `nvidia-text` alias, mapped to +`nvidia/nemotron-3-nano-30b-a3b`. Mimir now has `NVIDIA_API_KEY` in its Data Designer env file. +Use `MIMIR_SDG_MODEL_ALIAS=openrouter-text` only as a fallback. Local llama.cpp on port 8081 is +configured but not currently running. --- @@ -42,9 +43,12 @@ Step 1 (local — this project): │ Step 2 (remote — Mimir, optional): │ SCP seed JSONL + DD config to Mimir ◄────────┘ - SSH: data-designer create module_controller.yaml + SSH: source Data Designer env, then data-designer create module_controller.py --num-records --dataset-name module_controller_intents_ + --artifact-path /mnt/storage/data-designer/managed-assets/module_controller_artifacts_ + --output-format jsonl + --no-tui SCP expanded dataset back │ Step 3 (local — Zoe custom-local-llm): │ @@ -108,51 +112,65 @@ For each action in `actions.index.json`: Split: 90% train / 10% validation, partitioned by action so all 13 actions appear in both splits. +The deterministic seed files are intentionally small. They are good for validating the action +surface and dispatch schema, but they are not diverse enough for serious payload extraction +training by themselves. Use the Mimir-expanded `expanded.jsonl` file when training a model +that needs to generalize across phrasing, payload values, and action families. + --- ## Mimir Data Designer Config (auto-generated by bridge) The bridge writes this config to Mimir before running generation: -```yaml -# auto-generated by generate_training_data.py — do not edit by hand +```python +# auto-generated by generate_training_data.py - do not edit by hand # generated_at: -model_config_path: /mnt/storage/data-designer/model_configs.yaml -model_providers_path: /mnt/storage/data-designer/model_providers.yaml +from data_designer.config import DataDesignerConfigBuilder, LocalFileSeedSource -dataset: - name: module_controller_intents_ - schema_profile: instruction - seed_file: /mnt/storage/data-designer/managed-assets/module_controller_seed_.jsonl -columns: - - name: input - type: seed_passthrough +ACTION_IDS = ["calculate", "..."] - - name: output - type: llm_text - model_alias: nvidia-text - prompt: | - You are generating training data for a local action dispatcher. - Given the natural language request below, produce valid JSON with - exactly two fields: "action_id" (string) and "payload" (object). - The action_id must be one of: {action_ids}. - Vary the phrasing of the input naturally but keep the output schema strict. - Request: {{input}} - output_schema: - type: object - required: [action_id, payload] - additionalProperties: false - properties: - action_id: - type: string - enum: {action_ids} - payload: - type: object +def load_config_builder(): + builder = DataDesignerConfigBuilder( + model_configs="/mnt/storage/data-designer/model_configs.yaml", + ) + builder.with_seed_dataset( + LocalFileSeedSource( + path="/mnt/storage/data-designer/managed-assets/module_controller_seed_.jsonl" + ) + ) + builder.add_column( + name="expanded_record", + column_type="llm-structured", + model_alias="nvidia-text", + prompt="Generate one new strict instruction dataset row from the seed input/output.", + output_format={ + "type": "object", + "required": ["instruction", "input", "output", "metadata"], + "properties": { + "instruction": {"type": "string"}, + "input": {"type": "string"}, + "output": {"type": "string"}, + "metadata": { + "type": "object", + "required": ["action_id", "source", "split", "generated_at"], + "properties": { + "action_id": {"type": "string", "enum": ACTION_IDS} + }, + }, + }, + }, + ) + return builder ``` +Data Designer writes rows that include the generated `expanded_record` column. The bridge +copies that raw file back as `expanded.raw.jsonl`, extracts valid instruction rows, and writes +the model-training file as `expanded.jsonl`. + --- ## Manifest Entry (auto-patched by bridge) @@ -173,7 +191,9 @@ columns: "generator": "mp-ai-module-controller/scripts/generate_training_data.py", "sdg_engine": "data-designer", "sdg_host": "mimir (100.80.52.47)", - "sdg_model": "nvidia/nemotron-3-nano-30b-a3b", + "sdg_model": "nvidia-text", + "sdg_model_alias": "nvidia-text", + "expanded_path": "/Volumes/Zoe/custom-local-llm/training-data/module_controller_intents_/expanded.jsonl", "codebook_version": "1.0.0", "codebook_checksum": "", "registry_checksum": "", @@ -203,13 +223,14 @@ python3 scripts/generate_training_data.py \ --sdg-host 100.80.52.47 \ --sdg-user aaron-pressey \ --sdg-key ~/.ssh/silma_orson_ed25519 \ + --sdg-model-alias nvidia-text \ --num-records 1000 # Dry run (prints what would be generated, writes nothing): python3 scripts/generate_training_data.py --dry-run # Woodpecker deploy target: -scripts/deploy_training_data.sh +ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh ``` --- @@ -218,7 +239,8 @@ scripts/deploy_training_data.sh - Python 3.11+ (stdlib only: `json`, `pathlib`, `hashlib`, `time`, `argparse`, `subprocess`) - SSH access to Mimir via `~/.ssh/silma_orson_ed25519` (for NeMo expansion pass) -- `NVIDIA_API_KEY` set in Mimir's environment (for cloud Nemotron generation) +- `/home/aaron-pressey/.config/home-grown-llm-data/data-designer.env` present on Mimir with provider credentials +- Data Designer provider available for the selected alias. Current default: `nvidia-text`. - `dictionary/` files current — run `cargo run -- dictionary generate` first if registry changed --- diff --git a/llm.txt b/llm.txt index 38d04e3..c6235cb 100644 --- a/llm.txt +++ b/llm.txt @@ -33,6 +33,7 @@ Commands: - `scripts/precommit-check.sh` - `scripts/generate_training_data.py --dry-run` - `scripts/deploy_training_data.sh` +- `ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh` - `git config core.hooksPath .githooks` - `.codex/environments/environment.toml` @@ -60,4 +61,4 @@ Codex environment: `.codex/environments/environment.toml` defines setup plus Bui CI and hooks: `.gitea/workflows/ci.yml`, `.woodpecker.yml`, and `.githooks/pre-commit` all use `scripts/precommit-check.sh` to verify formatting, tests, and dictionary generation. -SDG bridge: `scripts/generate_training_data.py` reads `dictionary/actions.index.json`, `dictionary/model.codebook.json`, and `dictionary/model.examples.jsonl`, then writes instruction-format datasets to `/Volumes/Zoe/custom-local-llm/training-data/module_controller_intents_/` and patches `/Volumes/Zoe/custom-local-llm/manifest.llm.json`. Use `--dry-run` to validate without writes. Optional Mimir expansion requires `--sdg-host`, `--sdg-user`, and `--sdg-key`; failed remote expansion leaves local seeds and writes `SDG_ERROR.txt`. Woodpecker deploys the export with `scripts/deploy_training_data.sh`. +SDG bridge: `scripts/generate_training_data.py` reads `dictionary/actions.index.json`, `dictionary/model.codebook.json`, and `dictionary/model.examples.jsonl`, then writes instruction-format datasets to `/Volumes/Zoe/custom-local-llm/training-data/module_controller_intents_/` and patches `/Volumes/Zoe/custom-local-llm/manifest.llm.json`. Use `--dry-run` to validate without writes. Seed-only `train.jsonl` and `validation.jsonl` are useful for routing checks but need enrichment for serious training. Optional Mimir expansion requires `--sdg-host`, `--sdg-user`, and `--sdg-key`; it runs Data Designer over SSH, sources Mimir's Data Designer env file, uses `--sdg-model-alias` default `nvidia-text`, copies back `expanded.raw.jsonl`, and normalizes it to `expanded.jsonl`. Failed remote expansion leaves local seeds and writes `SDG_ERROR.txt`. Woodpecker deploys enriched data with `ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh`. diff --git a/manifest.llm.json b/manifest.llm.json index 018b67b..c11261e 100644 --- a/manifest.llm.json +++ b/manifest.llm.json @@ -30,6 +30,7 @@ "training_data_dry_run": "python3 scripts/generate_training_data.py --dry-run", "generate_training_data": "python3 scripts/generate_training_data.py", "deploy_training_data": "scripts/deploy_training_data.sh", + "deploy_training_data_with_mimir": "ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh", "install_hooks": "git config core.hooksPath .githooks" }, "environment": { @@ -73,6 +74,54 @@ "secret": false, "default": "info", "description": "Tracing log level for the controller." + }, + { + "name": "ENABLE_MIMIR_SDG", + "required": false, + "secret": false, + "default": "0", + "description": "Set to 1 to run Mimir/Data Designer expansion during training-data deploys." + }, + { + "name": "MIMIR_SDG_HOST", + "required": false, + "secret": false, + "default": "100.80.52.47", + "description": "Mimir host used for Data Designer expansion." + }, + { + "name": "MIMIR_SDG_USER", + "required": false, + "secret": false, + "default": "aaron-pressey", + "description": "SSH user for Mimir Data Designer expansion." + }, + { + "name": "MIMIR_SDG_KEY", + "required": false, + "secret": false, + "default": "~/.ssh/silma_orson_ed25519", + "description": "SSH key for Mimir Data Designer expansion." + }, + { + "name": "MIMIR_SDG_MODEL_ALIAS", + "required": false, + "secret": false, + "default": "nvidia-text", + "description": "Data Designer model alias used for SDG expansion." + }, + { + "name": "MIMIR_SDG_NUM_RECORDS", + "required": false, + "secret": false, + "default": "1000", + "description": "Number of enriched SDG records requested from Data Designer." + }, + { + "name": "NVIDIA_API_KEY", + "required": false, + "secret": true, + "description": "Provider credential for local Data Designer usage. Keep it in ignored .env files or Mimir's Data Designer env file." } ] }, @@ -239,7 +288,15 @@ "schema_profile": "instruction", "dry_run": "python3 scripts/generate_training_data.py --dry-run", "optional_sdg_host": "mimir (100.80.52.47)", - "sdg_engine": "data-designer" + "sdg_engine": "data-designer", + "default_sdg_model_alias": "nvidia-text", + "sdg_model_alias_override_env": "MIMIR_SDG_MODEL_ALIAS", + "sdg_invocation": "ssh to Mimir and run Data Designer CLI", + "sdg_env_file": "/home/aaron-pressey/.config/home-grown-llm-data/data-designer.env", + "expanded_raw_file": "expanded.raw.jsonl", + "expanded_normalized_file": "expanded.jsonl", + "woodpecker_deploy_command": "ENABLE_MIMIR_SDG=1 scripts/deploy_training_data.sh", + "seed_data_note": "train.jsonl and validation.jsonl are deterministic seeds; use expanded.jsonl for enriched model training when available." }, "ci": { "codex_environment": ".codex/environments/environment.toml", diff --git a/scripts/deploy_training_data.sh b/scripts/deploy_training_data.sh index 8d8b858..7dbad57 100755 --- a/scripts/deploy_training_data.sh +++ b/scripts/deploy_training_data.sh @@ -5,9 +5,22 @@ cd "$(dirname "$0")/.." export COPYFILE_DISABLE=1 +if [[ -f .env ]]; then + set -a + # shellcheck disable=SC1091 + source .env + set +a +fi + CUSTOM_LOCAL_LLM_ROOT="${CUSTOM_LOCAL_LLM_ROOT:-/Volumes/Zoe/custom-local-llm}" TRAINING_DATA_DIR="${TRAINING_DATA_DIR:-$CUSTOM_LOCAL_LLM_ROOT/training-data}" TRAINING_DATA_MANIFEST="${TRAINING_DATA_MANIFEST:-$CUSTOM_LOCAL_LLM_ROOT/manifest.llm.json}" +ENABLE_MIMIR_SDG="${ENABLE_MIMIR_SDG:-0}" +MIMIR_SDG_HOST="${MIMIR_SDG_HOST:-100.80.52.47}" +MIMIR_SDG_USER="${MIMIR_SDG_USER:-aaron-pressey}" +MIMIR_SDG_KEY="${MIMIR_SDG_KEY:-$HOME/.ssh/silma_orson_ed25519}" +MIMIR_SDG_MODEL_ALIAS="${MIMIR_SDG_MODEL_ALIAS:-nvidia-text}" +MIMIR_SDG_NUM_RECORDS="${MIMIR_SDG_NUM_RECORDS:-1000}" if [[ ! -d "$CUSTOM_LOCAL_LLM_ROOT" ]]; then echo "missing custom local LLM root: $CUSTOM_LOCAL_LLM_ROOT" >&2 @@ -22,10 +35,28 @@ if [[ ! -s "$TRAINING_DATA_MANIFEST" ]]; then fi cargo run -- dictionary generate -python3 scripts/generate_training_data.py \ +bridge_args=( + scripts/generate_training_data.py --dictionary-dir dictionary/ \ --output-dir "$TRAINING_DATA_DIR" \ --manifest "$TRAINING_DATA_MANIFEST" +) + +if [[ "$ENABLE_MIMIR_SDG" = "1" ]]; then + if [[ ! -r "$MIMIR_SDG_KEY" ]]; then + echo "missing Mimir SDG SSH key: $MIMIR_SDG_KEY" >&2 + exit 1 + fi + bridge_args+=( + --sdg-host "$MIMIR_SDG_HOST" + --sdg-user "$MIMIR_SDG_USER" + --sdg-key "$MIMIR_SDG_KEY" + --sdg-model-alias "$MIMIR_SDG_MODEL_ALIAS" + --num-records "$MIMIR_SDG_NUM_RECORDS" + ) +fi + +python3 "${bridge_args[@]}" python3 - <<'PY' import os diff --git a/scripts/generate_training_data.py b/scripts/generate_training_data.py index e1c834c..ddf0d79 100755 --- a/scripts/generate_training_data.py +++ b/scripts/generate_training_data.py @@ -5,6 +5,7 @@ from __future__ import annotations import argparse import json +import shlex import subprocess import sys import tempfile @@ -18,6 +19,8 @@ DEFAULT_MIN_RECORDS_PER_ACTION = 20 CUSTOM_LOCAL_LLM_ROOT = Path("/Volumes/Zoe/custom-local-llm") DEFAULT_OUTPUT_DIR = CUSTOM_LOCAL_LLM_ROOT / "training-data" DEFAULT_MANIFEST = CUSTOM_LOCAL_LLM_ROOT / "manifest.llm.json" +MIMIR_DATA_DESIGNER_BIN = "/home/aaron-pressey/.venvs/data-designer/bin/data-designer" +MIMIR_DATA_DESIGNER_ENV = "/home/aaron-pressey/.config/home-grown-llm-data/data-designer.env" def main() -> int: @@ -41,6 +44,7 @@ def parse_args() -> argparse.Namespace: parser.add_argument("--sdg-host") parser.add_argument("--sdg-user") parser.add_argument("--sdg-key") + parser.add_argument("--sdg-model-alias", default="nvidia-text") parser.add_argument("--num-records", type=int, default=1000) parser.add_argument( "--timestamp", @@ -71,6 +75,7 @@ def run_bridge(args: argparse.Namespace) -> dict[str, Any]: "output_dir": str(output_root / dataset_key), "manifest": str(manifest_path), "sdg_requested": has_sdg_args(args), + "sdg_model_alias": args.sdg_model_alias, } if args.dry_run: @@ -93,6 +98,7 @@ def run_bridge(args: argparse.Namespace) -> dict[str, Any]: generated_at=generated_at, seed_records=train_records + validation_records, action_ids=sorted(artifacts["actions"].keys()), + model_alias=args.sdg_model_alias, ) except Exception as exc: # noqa: BLE001 - keep local seed output and report remote failure. sdg_error = str(exc) @@ -108,6 +114,7 @@ def run_bridge(args: argparse.Namespace) -> dict[str, Any]: artifacts=artifacts, expanded_path=expanded_path, sdg_requested=has_sdg_args(args), + sdg_model_alias=args.sdg_model_alias, sdg_error=sdg_error, ) summary["expanded_path"] = str(expanded_path) if expanded_path else None @@ -339,19 +346,23 @@ def run_mimir_expansion( generated_at: int, seed_records: list[dict[str, Any]], action_ids: list[str], + model_alias: str, ) -> Path: require_sdg_args(args) remote_base = "/mnt/storage/data-designer/managed-assets" + remote_artifact_path = f"{remote_base}/module_controller_artifacts_{generated_at}" remote_seed = f"{remote_base}/module_controller_seed_{generated_at}.jsonl" - remote_config = f"{remote_base}/module_controller_{generated_at}.yaml" + remote_config = f"{remote_base}/module_controller_{generated_at}.py" + remote_expanded = f"{remote_artifact_path}/{dataset_key}.jsonl" expanded_path = dataset_dir / "expanded.jsonl" + raw_expanded_path = dataset_dir / "expanded.raw.jsonl" with tempfile.TemporaryDirectory() as tmp: tmp_path = Path(tmp) local_seed = tmp_path / "seed.jsonl" - local_config = tmp_path / "module_controller.yaml" + local_config = tmp_path / "module_controller.py" write_jsonl(local_seed, seed_records) - local_config.write_text(data_designer_config(dataset_key, generated_at, remote_seed, action_ids)) + local_config.write_text(data_designer_config(dataset_key, generated_at, remote_seed, action_ids, model_alias)) target = f"{args.sdg_user}@{args.sdg_host}" scp_base = ["scp", "-i", str(Path(args.sdg_key).expanduser())] @@ -361,59 +372,130 @@ def run_mimir_expansion( subprocess.run( [ *ssh_base, - "/home/aaron-pressey/.venvs/data-designer/bin/data-designer", - "create", - remote_config, - "--num-records", - str(args.num_records), - "--dataset-name", - dataset_key, + remote_bash_command( + remote_data_designer_create_command( + remote_config=remote_config, + num_records=args.num_records, + dataset_key=dataset_key, + artifact_path=remote_artifact_path, + ) + ), ], check=True, ) - subprocess.run([*scp_base, f"{target}:{remote_base}/{dataset_key}.jsonl", str(expanded_path)], check=True) + subprocess.run([*scp_base, f"{target}:{remote_expanded}", str(raw_expanded_path)], check=True) + normalize_expanded_jsonl(raw_expanded_path, expanded_path) return expanded_path -def data_designer_config(dataset_key: str, generated_at: int, remote_seed: str, action_ids: list[str]) -> str: - action_ids_json = json.dumps(action_ids) - return f"""# auto-generated by generate_training_data.py - do not edit by hand +def remote_data_designer_create_command( + remote_config: str, + num_records: int, + dataset_key: str, + artifact_path: str, +) -> str: + command = [ + MIMIR_DATA_DESIGNER_BIN, + "create", + remote_config, + "--num-records", + str(num_records), + "--dataset-name", + dataset_key, + "--artifact-path", + artifact_path, + "--output-format", + "jsonl", + "--no-tui", + ] + quoted_command = " ".join(shlex.quote(part) for part in command) + quoted_env = shlex.quote(MIMIR_DATA_DESIGNER_ENV) + return f"set -a; [ ! -f {quoted_env} ] || . {quoted_env}; set +a; {quoted_command}" + + +def remote_bash_command(command: str) -> str: + return f"bash -lc {shlex.quote(command)}" + + +def data_designer_config( + dataset_key: str, + generated_at: int, + remote_seed: str, + action_ids: list[str], + model_alias: str = "nvidia-text", +) -> str: + action_ids_repr = repr(action_ids) + action_ids_json = json.dumps(action_ids, separators=(",", ":")) + return f'''# auto-generated by generate_training_data.py - do not edit by hand # generated_at: {generated_at} -model_config_path: /mnt/storage/data-designer/model_configs.yaml -model_providers_path: /mnt/storage/data-designer/model_providers.yaml +from data_designer.config import DataDesignerConfigBuilder, LocalFileSeedSource -dataset: - name: {dataset_key} - schema_profile: instruction - seed_file: {remote_seed} -columns: - - name: input - type: seed_passthrough +ACTION_IDS = {action_ids_repr} - - name: output - type: llm_text - model_alias: nvidia-text - prompt: | - You are generating training data for a local action dispatcher. - Given the natural language request below, produce valid JSON with - exactly two fields: "action_id" (string) and "payload" (object). - The action_id must be one of: {action_ids_json}. - Vary the phrasing of the input naturally but keep the output schema strict. - Request: {{{{input}}}} - output_schema: - type: object - required: [action_id, payload] - additionalProperties: false - properties: - action_id: - type: string - enum: {action_ids_json} - payload: - type: object -""" +def load_config_builder(): + builder = DataDesignerConfigBuilder( + model_configs="/mnt/storage/data-designer/model_configs.yaml", + ) + builder.with_seed_dataset(LocalFileSeedSource(path="{remote_seed}")) + builder.add_column( + name="expanded_record", + column_type="llm-structured", + model_alias="{model_alias}", + prompt="""You are generating additional instruction-tuning rows for a local action dispatcher. + +Given this seed natural-language request and target JSON, create one new varied request +that maps to the same action_id. Keep the output schema strict. + +Allowed action_id values: {action_ids_json} + +Seed request: +{{{{ input }}}} + +Seed output JSON: +{{{{ output }}}} + +Return a complete instruction dataset row. The output field must be a JSON string with +exactly action_id and payload. The metadata.action_id must match that output action_id.""", + output_format={{ + "type": "object", + "required": ["instruction", "input", "output", "metadata"], + "additionalProperties": False, + "properties": {{ + "instruction": {{"type": "string", "minLength": 1}}, + "input": {{"type": "string", "minLength": 1}}, + "output": {{"type": "string", "minLength": 1}}, + "metadata": {{ + "type": "object", + "required": ["action_id", "source", "split", "generated_at"], + "additionalProperties": True, + "properties": {{ + "action_id": {{"type": "string", "enum": ACTION_IDS}}, + "source": {{"type": "string"}}, + "split": {{"type": "string"}}, + "generated_at": {{"type": "integer"}}, + }}, + }}, + }}, + }}, + ) + return builder +''' + + +def normalize_expanded_jsonl(raw_path: Path, output_path: Path) -> None: + normalized: list[dict[str, Any]] = [] + for row in read_jsonl(raw_path): + candidate = row.get("expanded_record") if isinstance(row, dict) else None + if isinstance(candidate, dict): + normalized.append(candidate) + elif isinstance(row, dict) and {"instruction", "input", "output", "metadata"}.issubset(row): + normalized.append(row) + if not normalized: + raise ValueError(f"no expanded instruction rows found in {raw_path}") + write_jsonl(output_path, normalized) def has_sdg_args(args: argparse.Namespace) -> bool: @@ -434,6 +516,7 @@ def patch_manifest( artifacts: dict[str, Any], expanded_path: Path | None, sdg_requested: bool, + sdg_model_alias: str | None, sdg_error: str | None = None, ) -> None: manifest = read_json(manifest_path) @@ -458,7 +541,8 @@ def patch_manifest( "generator": "mp-ai-module-controller/scripts/generate_training_data.py", "sdg_engine": "data-designer" if sdg_requested else None, "sdg_host": "mimir (100.80.52.47)" if sdg_requested else None, - "sdg_model": "nvidia/nemotron-3-nano-30b-a3b" if sdg_requested else None, + "sdg_model": sdg_model_alias if sdg_requested else None, + "sdg_model_alias": sdg_model_alias if sdg_requested else None, "expanded_path": str(expanded_path) if expanded_path else None, "sdg_error": sdg_error, "codebook_version": codebook["dictionary_version"], diff --git a/tests/test_generate_training_data.py b/tests/test_generate_training_data.py index 53ca5df..41bb758 100644 --- a/tests/test_generate_training_data.py +++ b/tests/test_generate_training_data.py @@ -49,6 +49,7 @@ class GenerateTrainingDataTests(unittest.TestCase): "sdg_host": None, "sdg_user": None, "sdg_key": None, + "sdg_model_alias": "nvidia-text", "num_records": 1000, "timestamp": 1234567890, } @@ -120,6 +121,58 @@ class GenerateTrainingDataTests(unittest.TestCase): "training-data/module_controller_intents_123", ) + def test_data_designer_config_declares_structured_expansion(self): + config = bridge.data_designer_config( + dataset_key="module_controller_intents_123", + generated_at=123, + remote_seed="/tmp/seed.jsonl", + action_ids=["logger"], + model_alias="nvidia-text", + ) + self.assertIn("load_config_builder", config) + self.assertIn('column_type="llm-structured"', config) + self.assertIn('model_alias="nvidia-text"', config) + self.assertIn("expanded_record", config) + + def test_remote_data_designer_command_sources_mimir_env(self): + command = bridge.remote_data_designer_create_command( + remote_config="/tmp/controller.py", + num_records=42, + dataset_key="module_controller_intents_123", + artifact_path="/tmp/artifacts", + ) + self.assertIn(bridge.MIMIR_DATA_DESIGNER_ENV, command) + self.assertIn(bridge.MIMIR_DATA_DESIGNER_BIN, command) + self.assertIn("--num-records 42", command) + self.assertIn("--no-tui", command) + + def test_remote_bash_command_quotes_inner_command(self): + command = bridge.remote_bash_command("set -a; echo ready") + self.assertTrue(command.startswith("bash -lc ")) + self.assertIn("'set -a; echo ready'", command) + + def test_normalize_expanded_jsonl_extracts_expanded_record(self): + raw = self.root / "expanded.raw.jsonl" + normalized = self.root / "expanded.jsonl" + source_row = { + "expanded_record": { + "instruction": bridge.INSTRUCTION, + "input": "record this message", + "output": json.dumps({"action_id": "logger", "payload": {"message": "hello"}}), + "metadata": { + "action_id": "logger", + "source": "data_designer", + "split": "train", + "generated_at": 123, + }, + } + } + bridge.write_jsonl(raw, [source_row]) + bridge.normalize_expanded_jsonl(raw, normalized) + rows = bridge.read_jsonl(normalized) + self.assertEqual(len(rows), 1) + self.assertEqual(rows[0]["metadata"]["action_id"], "logger") + if __name__ == "__main__": unittest.main()