From ea089ab634effe5cefb3e98e95e2aa99e40c837e Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Wed, 22 Jul 2026 22:18:46 +0800 Subject: [PATCH 01/33] megatron: infer MTP layers from model config --- .../Megatron-SWIFT/Command-line-parameters.md | 3 +- .../Megatron-SWIFT/Command-line-parameters.md | 3 +- swift/megatron/model/utils.py | 14 ++++- tests/megatron/test_model_config.py | 63 +++++++++++++++++++ 4 files changed, 78 insertions(+), 5 deletions(-) create mode 100644 tests/megatron/test_model_config.py diff --git a/docs/source/Megatron-SWIFT/Command-line-parameters.md b/docs/source/Megatron-SWIFT/Command-line-parameters.md index f4dc5c528b..9932b051b7 100644 --- a/docs/source/Megatron-SWIFT/Command-line-parameters.md +++ b/docs/source/Megatron-SWIFT/Command-line-parameters.md @@ -224,8 +224,7 @@ mHC 模块以在支持的 GPU 上获得更好的性能。需要安装 cuTile; - mhc_recompute_layer_num: 每个 MHC 重计算块的层数。设置后,每 `mhc_recompute_layer_num` 层构成一个重计算块。若为 None,Transformer 块中的所有层共享单个重计算块。默认为None。 **MTP参数** -- mtp_num_layers: 多token预测(MTP)层的数量。MTP将每个位置的预测范围扩展到多个未来token。此MTP实现使用D个顺序模块依次预测D个额外的token。默认为None。 - - 注意:mtp_num_layers的值,将不自动从config.json获取,需手动设置。你可以参考config.json中的`num_nextn_predict_layers`, `mtp_num_hidden_layers`字段填写该值。使用mcore-bridge时,将优先从safetensors文件中加载MTP权重,若无法找到,则进行随机初始化。 +- mtp_num_layers: 多token预测(MTP)层的数量。MTP将每个位置的预测范围扩展到多个未来token。此MTP实现使用D个顺序模块依次预测D个额外的token。默认从config.json中的`num_nextn_predict_layers`或`mtp_num_hidden_layers`读取;命令行显式设置时优先使用命令行值。使用mcore-bridge时,将优先从safetensors文件中加载MTP权重,若无法找到,则进行随机初始化。 - mtp_loss_scaling_factor: 多token预测(MTP)损失的缩放因子。我们计算所有深度上MTP损失的平均值,然后乘以该缩放因子得到总体MTP损失,它将作为一个额外的训练目标。默认为0.1。 - mtp_decoder_input_detach: 用来控制 MTP 分支里的 decoder_input 是否停止梯度。默认为False。开启后,MTP loss 不会直接通过 decoder_input 回传到 embedding/vit,但仍会通过 hidden_states 路径更新主干。 - mtp_shared_weights: MTP层之间共享权重,采用GLM-5使用的mtp方案。默认为False。例如你可以设置`--mtp_num_layers 3 --mtp_shared_weights true`。 diff --git a/docs/source_en/Megatron-SWIFT/Command-line-parameters.md b/docs/source_en/Megatron-SWIFT/Command-line-parameters.md index 4b331d6b1f..529c68203e 100644 --- a/docs/source_en/Megatron-SWIFT/Command-line-parameters.md +++ b/docs/source_en/Megatron-SWIFT/Command-line-parameters.md @@ -235,8 +235,7 @@ For guidance on selecting parallelization strategies, please refer to the [Train - mhc_recompute_layer_num: Number of layers per MHC recompute block. When set, every `mhc_recompute_layer_num` layers form a recompute block. If `None`, all layers in the transformer block share a single recompute block. Defaults to `None`. **MTP Parameters** -- mtp_num_layers: Number of Multi-Token Prediction (MTP) layers. MTP extends the prediction scope at each position to multiple future tokens. This MTP implementation uses D sequential modules to sequentially predict D additional tokens. Default is None. - - Note: The value of mtp_num_layers will not be automatically retrieved from config.json and must be set manually. You can refer to the `num_nextn_predict_layers`, `mtp_num_hidden_layers` field in config.json to fill in this value. When using mcore-bridge, MTP weights will be loaded from safetensors files first. If not found, random initialization will be performed. +- mtp_num_layers: Number of Multi-Token Prediction (MTP) layers. MTP extends the prediction scope at each position to multiple future tokens. This MTP implementation uses D sequential modules to sequentially predict D additional tokens. By default, the value is read from `num_nextn_predict_layers` or `mtp_num_hidden_layers` in config.json; an explicit command-line value takes precedence. When using mcore-bridge, MTP weights will be loaded from safetensors files first. If not found, random initialization will be performed. - mtp_loss_scaling_factor: Scaling factor of Multi-Token Prediction (MTP) loss. We compute the average of MTP losses across all depths, then multiply it by this scaling factor to obtain the overall MTP loss, which serves as an additional training objective. Default is 0.1. - mtp_decoder_input_detach: Controls whether to stop gradients through decoder_input in the MTP branch. Defaults to False. When enabled, the MTP loss will not back-propagate directly through decoder_input to the embedding/ViT, but will still update the backbone via the hidden_states pathway. - mtp_shared_weights: Share weights across MTP layers, following the MTP scheme proposed in GLM-5. Defaults to False. For example, you can set `--mtp_num_layers 3 --mtp_shared_weights true`. diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index acc2a7e48b..0cb3a57b62 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -5,7 +5,7 @@ from mcore_bridge import hf_to_mcore_config from transformers.utils import is_torch_npu_available -from swift.utils import get_logger +from swift.utils import HfConfigFactory, get_logger logger = get_logger() @@ -35,8 +35,20 @@ def _check_padding_free(args, config): args.padding_free = False +def _get_hf_mtp_num_layers(hf_config): + llm_config = HfConfigFactory.get_text_config(hf_config) + for key in ['num_nextn_predict_layers', 'mtp_num_hidden_layers']: + value = getattr(llm_config, key, None) + if value is not None: + return value + + def get_mcore_model_config(args, hf_config): kwargs = hf_to_mcore_config(hf_config) + if getattr(args, 'mtp_num_layers', None) is None: + mtp_num_layers = _get_hf_mtp_num_layers(hf_config) + if mtp_num_layers is not None: + kwargs['mtp_num_layers'] = mtp_num_layers kwargs['mcore_model_type'] = args.megatron_model_meta.model_type kwargs['hf_config'] = hf_config for f in fields(ModelConfig): diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py new file mode 100644 index 0000000000..31a9e21b2c --- /dev/null +++ b/tests/megatron/test_model_config.py @@ -0,0 +1,63 @@ +from types import SimpleNamespace + +import torch +from transformers import PretrainedConfig + +from swift.megatron.model import utils + + +class _ModelConfigStub: + + def __init__(self, **kwargs): + self.kwargs = kwargs + self.attention_backend = SimpleNamespace(name='unfused') + self.experimental_attention_variant = 'dsa' + + +def _make_args(mtp_num_layers=None): + return SimpleNamespace( + megatron_model_meta=SimpleNamespace(model_type='gpt'), + mtp_num_layers=mtp_num_layers, + task_type='causal_lm', + torch_dtype=torch.bfloat16, + decoder_first_pipeline_num_layers=None, + decoder_last_pipeline_num_layers=None, + fp4_param_gather=False, + fp8_param_gather=False, + moe_grouped_gemm=False, + router_replay_mode='disabled', + megatron_extra_kwargs=None, + padding_free=False, + ) + + +def _patch_model_config(monkeypatch): + monkeypatch.setattr(utils, 'ModelConfig', _ModelConfigStub) + monkeypatch.setattr(utils, 'fields', lambda _: [SimpleNamespace(name='mtp_num_layers')]) + + +def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): + _patch_model_config(monkeypatch) + hf_config = PretrainedConfig(num_nextn_predict_layers=1) + + config = utils.get_mcore_model_config(_make_args(), hf_config) + + assert config.kwargs['mtp_num_layers'] == 1 + + +def test_get_mcore_model_config_reads_mtp_num_hidden_layers(monkeypatch): + _patch_model_config(monkeypatch) + hf_config = PretrainedConfig(text_config=PretrainedConfig(mtp_num_hidden_layers=1)) + + config = utils.get_mcore_model_config(_make_args(), hf_config) + + assert config.kwargs['mtp_num_layers'] == 1 + + +def test_get_mcore_model_config_keeps_explicit_mtp_num_layers(monkeypatch): + _patch_model_config(monkeypatch) + hf_config = PretrainedConfig(num_nextn_predict_layers=1) + + config = utils.get_mcore_model_config(_make_args(mtp_num_layers=2), hf_config) + + assert config.kwargs['mtp_num_layers'] == 2 From 5499fddacffe66e953262f7ae57460f1e1256ff5 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Wed, 22 Jul 2026 22:45:49 +0800 Subject: [PATCH 02/33] megatron: prefer routed expert count from config --- swift/megatron/model/utils.py | 4 ++++ tests/megatron/test_model_config.py | 12 +++++++++++- 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index 0cb3a57b62..d05abc4868 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -45,6 +45,10 @@ def _get_hf_mtp_num_layers(hf_config): def get_mcore_model_config(args, hf_config): kwargs = hf_to_mcore_config(hf_config) + llm_config = HfConfigFactory.get_text_config(hf_config) + n_routed_experts = getattr(llm_config, 'n_routed_experts', None) + if n_routed_experts is not None: + kwargs['num_moe_experts'] = n_routed_experts if getattr(args, 'mtp_num_layers', None) is None: mtp_num_layers = _get_hf_mtp_num_layers(hf_config) if mtp_num_layers is not None: diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 31a9e21b2c..d6846f90e8 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -33,7 +33,8 @@ def _make_args(mtp_num_layers=None): def _patch_model_config(monkeypatch): monkeypatch.setattr(utils, 'ModelConfig', _ModelConfigStub) - monkeypatch.setattr(utils, 'fields', lambda _: [SimpleNamespace(name='mtp_num_layers')]) + monkeypatch.setattr( + utils, 'fields', lambda _: [SimpleNamespace(name='mtp_num_layers'), SimpleNamespace(name='num_moe_experts')]) def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): @@ -54,6 +55,15 @@ def test_get_mcore_model_config_reads_mtp_num_hidden_layers(monkeypatch): assert config.kwargs['mtp_num_layers'] == 1 +def test_get_mcore_model_config_prefers_n_routed_experts(monkeypatch): + _patch_model_config(monkeypatch) + hf_config = PretrainedConfig(num_experts=256, n_routed_experts=16) + + config = utils.get_mcore_model_config(_make_args(), hf_config) + + assert config.kwargs['num_moe_experts'] == 16 + + def test_get_mcore_model_config_keeps_explicit_mtp_num_layers(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) From a7754c3144022d7453d9dc922fb64879e0502390 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 23 Jul 2026 06:30:24 +0800 Subject: [PATCH 03/33] megatron:preserve-accuracy-compatible-DSA-norms --- swift/megatron/init.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/swift/megatron/init.py b/swift/megatron/init.py index c16e2b12f2..ef747b4dc9 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -120,7 +120,22 @@ def _patch_mcore_bridge(): require_version('mcore-bridge>=1.4.0', 'please install mcore-bridge via `pip install mcore-bridge -U`') import mcore_bridge from mcore_bridge import GPTBridge + from mcore_bridge.model.register import ModelLoader logger.info(f'mcore_bridge.__version__: {mcore_bridge.__version__}') + if not getattr(ModelLoader._replace_spec_dsa, '_swift_norm_accuracy_patch', False): + origin_replace_spec_dsa = ModelLoader._replace_spec_dsa + + def replace_spec_dsa(self, layer_spec): + origin_replace_spec_dsa(self, layer_spec) + if not getattr(self.config, 'norm_accuracy_compatible', False): + return + from megatron.core.transformer.torch_norm import AccuracyCompatibleRMSNorm + dsa_spec = layer_spec.submodules.self_attention + dsa_spec.submodules.q_layernorm = AccuracyCompatibleRMSNorm + dsa_spec.submodules.kv_layernorm = AccuracyCompatibleRMSNorm + + replace_spec_dsa._swift_norm_accuracy_patch = True + ModelLoader._replace_spec_dsa = replace_spec_dsa origin_save_weights = GPTBridge.save_weights def save_weights( From f1560a17b49129964b15acde1c50c73ef67284bd Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 20 Aug 2026 16:59:54 +0800 Subject: [PATCH 04/33] style: apply repo pre-commit (isort/yapf/single-quote) to files touched by this branch --- swift/megatron/init.py | 3 ++- swift/megatron/trainers/trainer.py | 6 +++--- tests/megatron/test_model_config.py | 8 ++++---- 3 files changed, 9 insertions(+), 8 deletions(-) diff --git a/swift/megatron/init.py b/swift/megatron/init.py index 5675537b49..b48ab537bd 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -10,10 +10,10 @@ from copy import copy, deepcopy from packaging import version from tqdm import tqdm -from typing import Optional from transformers.modeling_utils import custom_object_save from transformers.utils import is_torch_npu_available from transformers.utils.versions import require_version +from typing import Optional from swift.model import get_model_processor, save_checkpoint from swift.utils import (HfConfigFactory, disable_safe_ddp_context_use_barrier, get_logger, get_modules_to_not_convert, @@ -178,6 +178,7 @@ def _patch_mcore_bridge_disable_te(): import mcore_bridge.model.register as mcb_register def _force_local_spec(orig): + def wrapper(*args, **kwargs): kwargs['use_transformer_engine'] = False return orig(*args, **kwargs) diff --git a/swift/megatron/trainers/trainer.py b/swift/megatron/trainers/trainer.py index 7a1a261cb5..5579c73b04 100644 --- a/swift/megatron/trainers/trainer.py +++ b/swift/megatron/trainers/trainer.py @@ -78,9 +78,9 @@ def loss_func(self, import hashlib as _hashlib _final = (loss[0].detach().float() / loss[1].detach().float().clamp(min=1)).contiguous() print( - f"\nfinal_loss: rank={torch.distributed.get_rank()} " - f"val={_final.item():.20f} " - f"md5={_hashlib.md5(_final.cpu().numpy().tobytes()).hexdigest()}", + f'\nfinal_loss: rank={torch.distributed.get_rank()} ' + f'val={_final.item():.20f} ' + f'md5={_hashlib.md5(_final.cpu().numpy().tobytes()).hexdigest()}', flush=True) metrics = {'loss': reporting_loss} diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index d6846f90e8..70a5c579bd 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -1,7 +1,6 @@ -from types import SimpleNamespace - import torch from transformers import PretrainedConfig +from types import SimpleNamespace from swift.megatron.model import utils @@ -33,8 +32,9 @@ def _make_args(mtp_num_layers=None): def _patch_model_config(monkeypatch): monkeypatch.setattr(utils, 'ModelConfig', _ModelConfigStub) - monkeypatch.setattr( - utils, 'fields', lambda _: [SimpleNamespace(name='mtp_num_layers'), SimpleNamespace(name='num_moe_experts')]) + monkeypatch.setattr(utils, 'fields', + lambda _: [SimpleNamespace(name='mtp_num_layers'), + SimpleNamespace(name='num_moe_experts')]) def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): From cf3a6d0044dac23b4866f15811dba7239b8e2815 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Tue, 11 Aug 2026 17:12:00 +0800 Subject: [PATCH 05/33] feat(glm52): align Megatron training observability --- swift/arguments/base_args/base_args.py | 1 + swift/arguments/base_args/data_args.py | 8 + swift/dataset/__init__.py | 2 +- swift/dataset/loader.py | 22 +- swift/dataset/utils.py | 24 ++ swift/megatron/arguments/megatron_args.py | 1 + swift/megatron/callbacks/print.py | 18 + swift/megatron/init.py | 147 +++++--- swift/megatron/model/utils.py | 14 + swift/megatron/trainers/trainer.py | 315 ++++++++++++++++++ swift/megatron/trainers/utils.py | 7 +- swift/model/register.py | 12 +- swift/pipelines/train/sft.py | 10 +- tests/general/test_dataset_empty_assistant.py | 31 ++ tests/megatron/test_model_config.py | 35 ++ tests/megatron/test_pretokenized_dataset.py | 66 ++++ tests/megatron/test_raw_loss_observability.py | 72 ++++ 17 files changed, 730 insertions(+), 55 deletions(-) create mode 100644 tests/general/test_dataset_empty_assistant.py create mode 100644 tests/megatron/test_pretokenized_dataset.py create mode 100644 tests/megatron/test_raw_loss_observability.py diff --git a/swift/arguments/base_args/base_args.py b/swift/arguments/base_args/base_args.py index 326fb6d8b7..1be2f8aa5a 100644 --- a/swift/arguments/base_args/base_args.py +++ b/swift/arguments/base_args/base_args.py @@ -346,6 +346,7 @@ def get_model_processor(self, res['revision'] = revision or self.model_revision res['task_type'] = task_type or self.task_type res['num_labels'] = num_labels or self.num_labels + res['processor_id_or_path'] = getattr(self, 'tokenizer_name_or_path', None) return get_model_processor(**res) diff --git a/swift/arguments/base_args/data_args.py b/swift/arguments/base_args/data_args.py index 01b8b0f7a4..bb79cff2fb 100644 --- a/swift/arguments/base_args/data_args.py +++ b/swift/arguments/base_args/data_args.py @@ -58,6 +58,8 @@ class DataArguments: Example: '{"text1": "query", "text2": "response"}'. Defaults to None. strict (bool): If `True`, raises an error on any problematic data row. If `False`, discards the problematic sample and continues. Typically used for debugging. Defaults to False. + drop_empty_assistant_response (bool): If `True`, filters SFT rows whose assistant response is an empty string. + Defaults to False. remove_unused_columns (bool): Whether to remove columns not used by the model. If `False`, extra columns are passed to the trainer's `compute_loss` function, which is useful for custom loss calculations. Defaults to True. Note: The default is `False` for GPRO. @@ -76,6 +78,8 @@ class DataArguments: val_dataset: List[str] = field(default_factory=list) cached_dataset: List[str] = field(default_factory=list) cached_val_dataset: List[str] = field(default_factory=list) + pretokenized_dataset: bool = False + tokenizer_name_or_path: Optional[str] = None split_dataset_ratio: float = 0. data_seed: int = 42 @@ -91,6 +95,7 @@ class DataArguments: download_mode: Literal['force_redownload', 'reuse_dataset_if_exists'] = 'reuse_dataset_if_exists' columns: Optional[Union[dict, str]] = None strict: bool = False + drop_empty_assistant_response: bool = False remove_unused_columns: bool = True disable_auto_column_mapping: bool = False # Chinese name and English name @@ -118,6 +123,8 @@ def __post_init__(self): self._init_custom_dataset_info() if isinstance(self.cached_dataset, str): self.cached_dataset = [self.cached_dataset] + if self.pretokenized_dataset and not self.cached_dataset: + raise ValueError('pretokenized_dataset requires cached_dataset') self._init_val_dataset_exists() def _init_val_dataset_exists(self): @@ -138,6 +145,7 @@ def get_dataset_kwargs(self): 'download_mode': self.download_mode, 'columns': self.columns, 'strict': self.strict, + 'drop_empty_assistant_response': self.drop_empty_assistant_response, 'model_name': self.model_name, 'model_author': self.model_author, 'remove_unused_columns': self.remove_unused_columns, diff --git a/swift/dataset/__init__.py b/swift/dataset/__init__.py index baa76e0b62..876f2c00cc 100644 --- a/swift/dataset/__init__.py +++ b/swift/dataset/__init__.py @@ -11,7 +11,7 @@ from .register import (DATASET_MAPPING, DatasetMeta, SubsetDataset, get_dataset_list, register_dataset, register_dataset_info) from .utils import (AddLengthPreprocessor, EncodePreprocessor, LazyLLMDataset, get_temporary_cache_files_directory, - sample_dataset) + sample_dataset, validate_pretokenized_dataset) datasets.fingerprint.get_temporary_cache_files_directory = get_temporary_cache_files_directory datasets.arrow_dataset.get_temporary_cache_files_directory = get_temporary_cache_files_directory diff --git a/swift/dataset/loader.py b/swift/dataset/loader.py index 11f00d48d7..6e14aab563 100644 --- a/swift/dataset/loader.py +++ b/swift/dataset/loader.py @@ -28,6 +28,7 @@ def __init__( streaming: bool = False, hub_token: Optional[str] = None, strict: bool = False, + drop_empty_assistant_response: bool = False, download_mode: Literal['force_redownload', 'reuse_dataset_if_exists'] = 'reuse_dataset_if_exists', columns: Optional[Dict[str, str]] = None, remove_unused_columns: bool = True, @@ -38,11 +39,24 @@ def __init__( self.streaming = streaming self.hub_token = hub_token self.strict = strict + self.drop_empty_assistant_response = drop_empty_assistant_response self.download_mode = download_mode self.columns = columns self.remove_unused_columns = remove_unused_columns self.disable_auto_column_mapping = disable_auto_column_mapping + @staticmethod + def _has_nonempty_assistant_response(row: Dict) -> bool: + messages = row.get('messages') or [] + return all( + message.get('role') != 'assistant' or not isinstance(message.get('content'), str) + or bool(message['content'].strip()) for message in messages) + + def _filter_empty_assistant_responses(self, dataset: HfDataset) -> HfDataset: + if not self.drop_empty_assistant_response: + return dataset + return dataset.filter(self._has_nonempty_assistant_response) + def _load_dataset_path( self, dataset_path: str, @@ -66,7 +80,7 @@ def _load_dataset_path( enable_auto_mapping=not self.disable_auto_column_mapping) if self.remove_unused_columns: dataset = RowPreprocessor.remove_useless_columns(dataset) - return dataset + return self._filter_empty_assistant_responses(dataset) def _load_repo_dataset( self, @@ -136,7 +150,7 @@ def _load_repo_dataset( enable_auto_mapping=not self.disable_auto_column_mapping) if self.remove_unused_columns: dataset = RowPreprocessor.remove_useless_columns(dataset) - datasets.append(dataset) + datasets.append(self._filter_empty_assistant_responses(dataset)) return self.concat_datasets(datasets) @staticmethod @@ -236,6 +250,7 @@ def load_dataset( use_hf: Optional[bool] = None, hub_token: Optional[str] = None, strict: bool = False, + drop_empty_assistant_response: bool = False, download_mode: Literal['force_redownload', 'reuse_dataset_if_exists'] = 'reuse_dataset_if_exists', columns: Optional[Dict[str, str]] = None, # columns_mapping remove_unused_columns: bool = True, @@ -278,6 +293,8 @@ def load_dataset( hub_token: Authentication token for accessing private datasets on the hub. Default: None. strict: If True, raise exceptions when encountering malformed data rows. If False, skip invalid rows with warnings. Default: False. + drop_empty_assistant_response: Filter rows whose assistant response is an empty string. + Defaults to False. download_mode: How to handle existing cached datasets: - 'reuse_dataset_if_exists': Use cached version if available - 'force_redownload': Always download fresh copy @@ -342,6 +359,7 @@ def load_dataset( streaming=streaming, hub_token=hub_token, strict=strict, + drop_empty_assistant_response=drop_empty_assistant_response, download_mode=download_mode, columns=columns, # columns_mapping remove_unused_columns=remove_unused_columns, diff --git a/swift/dataset/utils.py b/swift/dataset/utils.py index e78abf12f2..23e2ac692a 100644 --- a/swift/dataset/utils.py +++ b/swift/dataset/utils.py @@ -54,6 +54,30 @@ def sample_dataset( return dataset +def validate_pretokenized_dataset(dataset: HfDataset, max_length: int) -> None: + """Validate fixed token rows without invoking a template or tokenizer.""" + required = {'input_ids', 'labels', 'position_ids', 'lengths'} + if not isinstance(dataset, HfDataset): + raise TypeError('pretokenized_dataset requires a Hugging Face Dataset') + missing = required.difference(dataset.column_names) + if missing: + raise ValueError(f'pretokenized dataset missing columns: {sorted(missing)}') + for row in dataset: + input_ids = row['input_ids'] + labels = row['labels'] + position_ids = row['position_ids'] + length_value = row['lengths'] + length = max(length_value) if isinstance(length_value, list) else length_value + if not isinstance(length, int) or length <= 0 or length > max_length: + raise ValueError(f'pretokenized dataset has invalid length: {length}') + if len(input_ids) != length or len(labels) != length: + raise ValueError('pretokenized dataset has inconsistent input/label length') + if len(position_ids) != length: + raise ValueError('pretokenized dataset has inconsistent position_ids length') + if any(not isinstance(token, int) for token in input_ids + labels + position_ids): + raise TypeError('pretokenized dataset fields must contain integer token values') + + class LazyLLMDataset(Dataset): """This class if used to lazy tokenize the dataset, and skips bad ones when training""" diff --git a/swift/megatron/arguments/megatron_args.py b/swift/megatron/arguments/megatron_args.py index d04331f4fe..bc60b9d7a3 100644 --- a/swift/megatron/arguments/megatron_args.py +++ b/swift/megatron/arguments/megatron_args.py @@ -549,6 +549,7 @@ class MegatronArguments(RLHFMegatronArgumentsMixin, MegatronTunerMixin): start_weight_decay: Optional[float] = None end_weight_decay: Optional[float] = None clip_grad: float = 1. + native_unfused_adamw: bool = False adam_beta1: float = 0.9 adam_beta2: float = 0.95 adam_eps: float = 1e-8 diff --git a/swift/megatron/callbacks/print.py b/swift/megatron/callbacks/print.py index 01a41b385d..83ec6f71c5 100644 --- a/swift/megatron/callbacks/print.py +++ b/swift/megatron/callbacks/print.py @@ -11,6 +11,16 @@ logger = get_logger() +def raw_loss_event(step, logs): + """Return an unrounded training-loss event, excluding evaluation metrics.""" + raw_losses = { + key: value + for key, value in logs.items() + if key == 'loss' or (key.startswith('mtp_') and key.endswith('_loss')) + } + return {'step': step, **raw_losses} if raw_losses else None + + class PrintCallback(MegatronCallback): def __init__(self, trainer): @@ -18,6 +28,7 @@ def __init__(self, trainer): self.training_bar = None self.eval_bar = None self.jsonl_writer = None + self.raw_loss_writer = None self.is_write_rank = is_last_rank() def on_train_begin(self): @@ -30,6 +41,10 @@ def on_train_begin(self): logging_path = os.path.join(self.args.output_dir, 'logging.jsonl') logger.info(f'logging_path: {logging_path}') self.jsonl_writer = JsonlWriter(logging_path, enable_async=True, write_on_rank='last') + raw_loss_path = os.environ.get('MODEL_REPRO_RAW_LOSS_PATH') + if raw_loss_path: + logger.info(f'raw_loss_path: {raw_loss_path}') + self.raw_loss_writer = JsonlWriter(raw_loss_path, write_on_rank='last') def on_train_end(self): self.training_bar.close() @@ -63,6 +78,9 @@ def on_log(self, logs): memory = reduce_max_stat_across_model_parallel_group(torch.cuda.max_memory_reserved() / 1024**3) logs['memory(GiB)'] = round(memory, 2) logs['train_speed(s/it)'] = round(train_speed, 6) + raw_event = raw_loss_event(state.iteration, logs) + if self.raw_loss_writer is not None and raw_event is not None: + self.raw_loss_writer.append(raw_event) logs = {k: round(v, 8) if isinstance(v, float) else v for k, v in logs.items()} self.jsonl_writer.append(logs) if self.is_write_rank: diff --git a/swift/megatron/init.py b/swift/megatron/init.py index b48ab537bd..ef294657a9 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -16,19 +16,31 @@ from typing import Optional from swift.model import get_model_processor, save_checkpoint -from swift.utils import (HfConfigFactory, disable_safe_ddp_context_use_barrier, get_logger, get_modules_to_not_convert, - get_multimodal_target_regex, is_master, split_list) +from swift.utils import ( + HfConfigFactory, + disable_safe_ddp_context_use_barrier, + get_logger, + get_modules_to_not_convert, + get_multimodal_target_regex, + is_master, + split_list, +) logger = get_logger() +def _get_save_processor_id(args): + """Use the configured processor source when weights and tokenizer are independent.""" + return getattr(args, "tokenizer_name_or_path", None) or args.model_dir + + def _patch__batched_p2p_ops(): from megatron.core.pipeline_parallel import p2p_communication _batched_p2p_ops_origin = p2p_communication._batched_p2p_ops def _batched_p2p_ops(**kwargs): - kwargs['group'] = None + kwargs["group"] = None return _batched_p2p_ops_origin(**kwargs) p2p_communication._batched_p2p_ops = _batched_p2p_ops @@ -37,9 +49,10 @@ def _batched_p2p_ops(**kwargs): def _patch_torch_FileSystemReader(): from torch.distributed.checkpoint.filesystem import FileSystemReader from torch.futures import Future + _origin_read_data = FileSystemReader.read_data _origin__slice_file = FileSystemReader._slice_file - READER_MAX_WORKERS = int(os.environ.get('MCORE_READER_MAX_WORKERS', '16')) + READER_MAX_WORKERS = int(os.environ.get("MCORE_READER_MAX_WORKERS", "16")) @contextmanager def _patch__slice_file(prog_bar): @@ -59,10 +72,12 @@ def read_data(self, plan, planner): def _worker(plan_shard): _origin_read_data(self, plan_shard, planner) - prog_bar = tqdm(total=len(plan.items), dynamic_ncols=True, desc='Loading: ') + prog_bar = tqdm(total=len(plan.items), dynamic_ncols=True, desc="Loading: ") plan_shards = split_list(plan.items, READER_MAX_WORKERS, contiguous=False) with _patch__slice_file(prog_bar): - with concurrent.futures.ThreadPoolExecutor(max_workers=READER_MAX_WORKERS) as pool: + with concurrent.futures.ThreadPoolExecutor( + max_workers=READER_MAX_WORKERS + ) as pool: futures = [] for i in range(READER_MAX_WORKERS): plan_shard = copy(plan) @@ -86,8 +101,12 @@ def _patch_validate_non_overlapping_shards_metadata(): def validate_non_overlapping_shards_metadata(*args, **kwargs): pass - api.validate_non_overlapping_shards_metadata = validate_non_overlapping_shards_metadata - api2.validate_non_overlapping_shards_metadata = validate_non_overlapping_shards_metadata + api.validate_non_overlapping_shards_metadata = ( + validate_non_overlapping_shards_metadata + ) + api2.validate_non_overlapping_shards_metadata = ( + validate_non_overlapping_shards_metadata + ) def _validate_global_plan(*args, **kwargs): # torch returns a list of error messages here; empty list means "no error". @@ -119,11 +138,12 @@ def _patch_unified_memory(): return from torch.utils import cpp_extension + load_inline = cpp_extension.load_inline def _new_load_inline(*args, **kwargs): - name = kwargs.get('name') - if name == 'managed_alloc_runtime': + name = kwargs.get("name") + if name == "managed_alloc_runtime": raise RuntimeError return load_inline(*args, **kwargs) @@ -226,24 +246,28 @@ def _set_layer_attn(self, mg_layer, hf_state_dict, layer_idx, to_mcore): def _patch_mcore_bridge(): - require_version('mcore-bridge>=1.4.0', 'please install mcore-bridge via `pip install mcore-bridge -U`') + require_version( + "mcore-bridge>=1.4.0", + "please install mcore-bridge via `pip install mcore-bridge -U`", + ) import mcore_bridge from mcore_bridge import GPTBridge from mcore_bridge.model.register import ModelLoader - logger.info(f'mcore_bridge.__version__: {mcore_bridge.__version__}') + logger.info(f"mcore_bridge.__version__: {mcore_bridge.__version__}") if _use_accuracy_compatible_enabled(): _patch_mcore_bridge_disable_te() - if not getattr(ModelLoader._replace_spec_dsa, '_swift_norm_accuracy_patch', False): + if not getattr(ModelLoader._replace_spec_dsa, "_swift_norm_accuracy_patch", False): origin_replace_spec_dsa = ModelLoader._replace_spec_dsa def replace_spec_dsa(self, layer_spec): origin_replace_spec_dsa(self, layer_spec) - if not getattr(self.config, 'norm_accuracy_compatible', False): + if not getattr(self.config, "norm_accuracy_compatible", False): return - from megatron.core.transformer.torch_norm import AccuracyCompatibleRMSNorm + from megatron.core.transformer.torch_norm import WrappedTorchNorm + dsa_spec = layer_spec.submodules.self_attention - dsa_spec.submodules.q_layernorm = AccuracyCompatibleRMSNorm - dsa_spec.submodules.kv_layernorm = AccuracyCompatibleRMSNorm + dsa_spec.submodules.q_layernorm = WrappedTorchNorm + dsa_spec.submodules.kv_layernorm = WrappedTorchNorm replace_spec_dsa._swift_norm_accuracy_patch = True ModelLoader._replace_spec_dsa = replace_spec_dsa @@ -254,41 +278,53 @@ def save_weights( mg_models, output_dir: str, peft_format: bool = False, - max_shard_size: str = '5GB', + max_shard_size: str = "5GB", args=None, processor=None, ) -> None: - origin_save_weights(self, mg_models, output_dir, peft_format=peft_format, max_shard_size=max_shard_size) + origin_save_weights( + self, + mg_models, + output_dir, + peft_format=peft_format, + max_shard_size=max_shard_size, + ) if processor is None or args is None: return hf_config = self.config.hf_config hf_config = deepcopy(hf_config) - if is_master() and not hasattr(self, 'hf_model'): - if hasattr(self, 'get_hf_meta_model'): + if is_master() and not hasattr(self, "hf_model"): + if hasattr(self, "get_hf_meta_model"): self.hf_model = self.get_hf_meta_model() self.hf_model.model_meta = processor.model_meta self.hf_model.model_info = processor.model_info else: - with torch.device('meta'), disable_safe_ddp_context_use_barrier(): + with torch.device("meta"), disable_safe_ddp_context_use_barrier(): self.hf_model = get_model_processor( - args.model_dir, model_type=args.model_type, return_dummy_model=True)[0] + args.model_dir, + model_type=args.model_type, + return_dummy_model=True, + processor_id_or_path=_get_save_processor_id(args), + )[0] if is_master(): if peft_format: peft_config = copy(mg_models[0].peft_config[self._adapter_name]) - if self.config.task_type == 'seq_cls': - peft_config.task_type = 'SEQ_CLS' - if self.is_multimodal and 'all-linear' in args.target_modules: + if self.config.task_type == "seq_cls": + peft_config.task_type = "SEQ_CLS" + if self.is_multimodal and "all-linear" in args.target_modules: peft_config.target_modules = get_multimodal_target_regex( self.hf_model, freeze_llm=args.freeze_llm, freeze_vit=args.freeze_vit, freeze_aligner=args.freeze_aligner, - include_embedding='all-embedding' in args.target_modules, - exclude_router='all-router' not in args.target_modules) + include_embedding="all-embedding" in args.target_modules, + exclude_router="all-router" not in args.target_modules, + ) else: assert not isinstance(peft_config.target_modules, str), ( - 'target_regex is not currently supported for LoRA conversion. Please set `--merge_lora true`.') + "target_regex is not currently supported for LoRA conversion. Please set `--merge_lora true`." + ) peft_config.target_modules = self._peft_target_modules peft_config.modules_to_save = self._peft_modules_to_save peft_config.save_pretrained(output_dir) @@ -296,43 +332,59 @@ def save_weights( config = self.config llm_config = HfConfigFactory.get_text_config(hf_config) if config.mtp_num_layers: - for key in ['num_nextn_predict_layers', 'mtp_num_hidden_layers']: + for key in ["num_nextn_predict_layers", "mtp_num_hidden_layers"]: if hasattr(llm_config, key): setattr(llm_config, key, config.mtp_num_layers) break else: llm_config.num_nextn_predict_layers = config.mtp_num_layers - HfConfigFactory.del_config_attr(hf_config, 'quantization_config') + HfConfigFactory.del_config_attr(hf_config, "quantization_config") expert_dtype = None - if config.fp8 is not None and config.fp8_recipe == 'blockwise' and config.fp8_param: - from transformers.utils.quantization_config import FineGrainedFP8Config + if ( + config.fp8 is not None + and config.fp8_recipe == "blockwise" + and config.fp8_param + ): + from transformers.utils.quantization_config import ( + FineGrainedFP8Config, + ) + modules_to_not_convert = get_modules_to_not_convert(self.hf_model) - if hasattr(self, '_fp8_skip_modules'): - modules_to_not_convert = (modules_to_not_convert or []) + list(self._fp8_skip_modules) - hf_config.quantization_config = FineGrainedFP8Config(modules_to_not_convert=modules_to_not_convert) - expert_dtype = 'fp8' - if args.model_type == 'deepseek_v4': - HfConfigFactory.set_config_attr(hf_config, 'expert_dtype', expert_dtype) + if hasattr(self, "_fp8_skip_modules"): + modules_to_not_convert = (modules_to_not_convert or []) + list( + self._fp8_skip_modules + ) + hf_config.quantization_config = FineGrainedFP8Config( + modules_to_not_convert=modules_to_not_convert + ) + expert_dtype = "fp8" + if args.model_type == "deepseek_v4": + HfConfigFactory.set_config_attr( + hf_config, "expert_dtype", expert_dtype + ) hf_config.save_pretrained(output_dir) - if getattr(self.hf_model, '_auto_class') is not None: + if getattr(self.hf_model, "_auto_class") is not None: try: custom_object_save(self.hf_model, output_dir, config=hf_config) except FileNotFoundError as e: - logger.error(f'custom_object_save Error: {e}') + logger.error(f"custom_object_save Error: {e}") save_checkpoint( None, processor, output_dir, model_dirs=[args.model_dir], - additional_saved_files=self.hf_model.model_meta.additional_saved_files) - logger.info(f'Successfully saved `safetensors` model weights in `{output_dir}`.') + additional_saved_files=self.hf_model.model_meta.additional_saved_files, + ) + logger.info( + f"Successfully saved `safetensors` model weights in `{output_dir}`." + ) dist.barrier() # Ensure all weights are saved completely GPTBridge.save_weights = save_weights def init_megatron_env(): - os.environ.pop('VLLM_USE_MODELSCOPE', None) + os.environ.pop("VLLM_USE_MODELSCOPE", None) logging_level = logging.root.level _patch_unified_memory() _patch_transformers_output_recorder() @@ -342,11 +394,12 @@ def init_megatron_env(): try: _patch_torch_FileSystemReader() except Exception: - logger.warning('Failed to patch FileSystemReader.') + logger.warning("Failed to patch FileSystemReader.") try: _patch_validate_non_overlapping_shards_metadata() except Exception: - logger.warning('Patch validate_non_overlapping_shards_metadata failed.') + logger.warning("Patch validate_non_overlapping_shards_metadata failed.") pass import megatron.core - logger.info(f'megatron.core.__version__: {megatron.core.__version__}') + + logger.info(f"megatron.core.__version__: {megatron.core.__version__}") diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index d05abc4868..f7b2c1147d 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -35,6 +35,19 @@ def _check_padding_free(args, config): args.padding_free = False +def _check_dsa_index_share_recompute(config): + """Reject activation replay that omits a DSA skip layer's source indexer.""" + if ( + config.experimental_attention_variant == 'dsa' + and (getattr(config, 'dsa_indexer_topk_freq', 1) or 1) > 1 + and getattr(config, 'recompute_granularity', None) not in {None, 'none'} + ): + raise ValueError( + 'DSA cross-layer top-k sharing is incompatible with activation recompute because a skip layer may be ' + 'replayed without its source computing layer. Set recompute_granularity=none.' + ) + + def _get_hf_mtp_num_layers(hf_config): llm_config = HfConfigFactory.get_text_config(hf_config) for key in ['num_nextn_predict_layers', 'mtp_num_hidden_layers']: @@ -93,6 +106,7 @@ def get_mcore_model_config(args, hf_config): setattr(config, 'use_flash_attn', True) _check_attention_backend(args, config) _check_padding_free(args, config) + _check_dsa_index_share_recompute(config) return config diff --git a/swift/megatron/trainers/trainer.py b/swift/megatron/trainers/trainer.py index 5579c73b04..13e22f297f 100644 --- a/swift/megatron/trainers/trainer.py +++ b/swift/megatron/trainers/trainer.py @@ -1,4 +1,7 @@ # Copyright (c) ModelScope Contributors. All rights reserved. +import hashlib +import json +import os import torch import torch.distributed as dist import torch.nn @@ -16,8 +19,318 @@ logger = get_logger() +def project_owning_loader_semantics(input_values, model_label_values, semantic_length, labels_were_shifted=True): + """Normalize padded Megatron carrier tensors back to the dataset semantic row.""" + semantic_length = int(semantic_length) + if semantic_length <= 0 or semantic_length > len(input_values) or semantic_length > len(model_label_values): + raise ValueError( + f'invalid owning-loader semantic length {semantic_length} for carrier lengths ' + f'{len(input_values)}/{len(model_label_values)}') + semantic_input_values = input_values[:semantic_length] + normalized_label_values = model_label_values + if labels_were_shifted and model_label_values: + # get_batch_on_this_pp_rank rolls causal-LM labels left by one before + # model forward. Reverse that roll for a framework-neutral dataset receipt. + normalized_label_values = model_label_values[-1:] + model_label_values[:-1] + semantic_label_values = normalized_label_values[:semantic_length] + semantic_mask_values = [label != -100 for label in semantic_label_values] + return semantic_input_values, semantic_label_values, semantic_mask_values + + class MegatronTrainer(BaseMegatronTrainer): + _LAYER0_FINE_FORWARD_MODULES = { + 'decoder.layers.0.input_layernorm': 'layer0_input_rmsnorm_output', + 'decoder.layers.0.self_attention.linear_q_down_proj': 'layer0_q_down_projection_output', + 'decoder.layers.0.self_attention.q_layernorm': 'layer0_q_rmsnorm_output', + 'decoder.layers.0.self_attention.linear_q_up_proj': 'layer0_q_up_projection_output', + 'decoder.layers.0.self_attention.linear_kv_down_proj': 'layer0_kv_down_projection_output', + 'decoder.layers.0.self_attention.kv_layernorm': 'layer0_kv_rmsnorm_output', + 'decoder.layers.0.self_attention.linear_kv_up_proj': 'layer0_kv_up_projection_output', + 'decoder.layers.0.self_attention.linear_proj': 'layer0_attention_output_projection', + 'decoder.layers.0.self_attention': 'layer0_self_attention_output', + 'decoder.layers.0.mlp.linear_fc1': 'layer0_dense_fc1_output', + 'decoder.layers.0.mlp.linear_fc2': 'layer0_dense_fc2_output', + 'decoder.layers.0.mlp': 'layer0_dense_mlp_output', + 'decoder.layers.0': 'base_transformer_layer_0_output', + } + + @classmethod + def _forward_contract_specs(cls, boundary_set): + if boundary_set == 'coarse': + return None + if boundary_set == 'layer0_fine': + return dict(cls._LAYER0_FINE_FORWARD_MODULES) + raise ValueError(f'unsupported MODEL_REPRO_FORWARD_BOUNDARY_SET: {boundary_set}') + + @staticmethod + def _parameter_record(param): + if hasattr(param, 'is_dist') and param.is_dist(): + param = param._local_value() + tensor = param.detach().contiguous().to(device='cpu') + + def raw_digest(value): + return hashlib.sha256(value.contiguous().view(torch.uint8).numpy().tobytes()).hexdigest() + + zero_count = 0 + negative_zero_count = 0 + if tensor.is_floating_point(): + zero_mask = tensor == 0 + zero_count = int(zero_mask.sum().item()) + negative_zero_count = int((zero_mask & torch.signbit(tensor)).sum().item()) + record = { + 'shape': list(tensor.shape), + 'dtype': str(tensor.dtype), + 'numel': tensor.numel(), + 'sha256': raw_digest(tensor), + 'positive_zero_count': zero_count - negative_zero_count, + 'negative_zero_count': negative_zero_count, + } + if tensor.ndim == 2: + record['transpose_sha256'] = raw_digest(tensor.transpose(0, 1)) + return record + + @staticmethod + def _first_tensor(value): + if isinstance(value, torch.Tensor): + return value + if isinstance(value, dict): + for item in value.values(): + tensor = MegatronTrainer._first_tensor(item) + if tensor is not None: + return tensor + if isinstance(value, (tuple, list)): + for item in value: + tensor = MegatronTrainer._first_tensor(item) + if tensor is not None: + return tensor + return None + + def _write_forward_record(self, boundary, value): + tensor = self._first_tensor(value) + if tensor is None: + return + output_dir = os.environ.get('MODEL_REPRO_FORWARD_RECEIPT_DIR') + rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 + rank_dir = os.path.join(output_dir, f'rank{rank}') + os.makedirs(rank_dir, exist_ok=True) + records = getattr(self, '_forward_contract_records', {}) + call_index = sum(name == boundary or name.startswith(f'{boundary}_call') for name in records) + name = boundary if call_index == 0 else f'{boundary}_call{call_index}' + tensor = tensor.detach().contiguous().to(device='cpu') + raw = tensor.view(torch.uint8).numpy().tobytes() + file_name = ''.join(character if character.isalnum() or character in '-_' else '_' for character in name) + raw_path = os.path.join(rank_dir, f'{file_name}.bin') + with open(raw_path, 'wb') as stream: + stream.write(raw) + zero_count = 0 + negative_zero_count = 0 + if tensor.is_floating_point(): + zero_mask = tensor == 0 + zero_count = int(zero_mask.sum().item()) + negative_zero_count = int((zero_mask & torch.signbit(tensor)).sum().item()) + records[name] = { + 'boundary': boundary, + 'shape': list(tensor.shape), + 'dtype': str(tensor.dtype), + 'numel': tensor.numel(), + 'sha256': hashlib.sha256(raw).hexdigest(), + 'positive_zero_count': zero_count - negative_zero_count, + 'negative_zero_count': negative_zero_count, + 'raw_path': raw_path, + } + self._forward_contract_records = records + payload = { + 'schema': 'glm52-local-forward-boundaries/v1', + 'framework': 'torch', + 'rank': rank, + 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, + 'boundary_set': getattr(self, '_forward_contract_boundary_set', 'coarse'), + 'selectors': getattr(self, '_forward_contract_selector_receipt', []), + 'records': records, + } + with open(os.path.join(rank_dir, 'metadata.json'), 'w', encoding='utf-8') as stream: + json.dump(payload, stream, ensure_ascii=False, indent=2, sort_keys=True) + stream.write('\n') + + def _install_forward_contract_once(self): + output_dir = os.environ.get('MODEL_REPRO_FORWARD_RECEIPT_DIR') + if not output_dir or getattr(self, '_forward_contract_installed', False): + return + rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 + boundary_set = os.environ.get('MODEL_REPRO_FORWARD_BOUNDARY_SET', 'coarse') + fine_specs = self._forward_contract_specs(boundary_set) + self._forward_contract_boundary_set = boundary_set + handles = [] + if fine_specs is not None: + selected = [] + if rank < 2: + module_hits = {name: [] for name in fine_specs} + for chunk_index, model in enumerate(self.unwrapped_models): + for module_name, module in model.named_modules(): + if module_name in module_hits: + module_hits[module_name].append((chunk_index, module)) + invalid = {name: len(hits) for name, hits in module_hits.items() if len(hits) != 1} + if invalid: + raise RuntimeError(f'layer0 fine forward selectors must match exactly once on rank {rank}: {invalid}') + for module_name, boundary in fine_specs.items(): + chunk_index, module = module_hits[module_name][0] + handles.append(module.register_forward_hook( + lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) + selected.append({'chunk': chunk_index, 'module': module_name, 'boundary': boundary}) + self._forward_contract_selector_receipt = selected + rank_dir = os.path.join(output_dir, f'rank{rank}') + os.makedirs(rank_dir, exist_ok=True) + with open(os.path.join(rank_dir, 'metadata.json'), 'w', encoding='utf-8') as stream: + json.dump({ + 'schema': 'glm52-local-forward-boundaries/v1', + 'framework': 'torch', + 'rank': rank, + 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, + 'boundary_set': boundary_set, + 'selectors': selected, + 'records': {}, + }, stream, ensure_ascii=False, indent=2, sort_keys=True) + stream.write('\n') + self._forward_contract_handles = handles + self._forward_contract_installed = True + return + for chunk_index, model in enumerate(self.unwrapped_models): + for module_name, module in model.named_modules(): + boundary = None + match = __import__('re').fullmatch(r'decoder\.layers\.(\d+)', module_name) + if module_name == 'embedding': + boundary = f'chunk{chunk_index}_embedding_output' + elif match: + local_layer = int(match.group(1)) + global_layer = local_layer if rank < 2 else local_layer + 2 + input_boundary = f'base_layer_{global_layer}_input' + handles.append(module.register_forward_pre_hook( + lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) + boundary = f'base_layer_{global_layer}_output' + elif module_name == 'decoder.final_layernorm': + boundary = 'final_norm_output' + elif module_name == 'output_layer': + input_boundary = 'output_head_input' + handles.append(module.register_forward_pre_hook( + lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) + boundary = 'output_head_output' + elif module_name.startswith('mtp.layers.0.') and module_name.rsplit('.', 1)[-1] in { + 'enorm', 'hnorm', 'eh_proj', 'mtp_model_layer', 'layer_norm', 'final_layernorm'}: + boundary = f"mtp_{module_name.removeprefix('mtp.layers.0.').replace('.', '_')}_output" + if boundary is not None: + handles.append(module.register_forward_hook( + lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) + self._forward_contract_handles = handles + self._forward_contract_installed = True + + def _write_parameter_contract_once(self): + output_dir = os.environ.get('MODEL_REPRO_PARAMETER_RECEIPT_DIR') + if not output_dir or getattr(self, '_parameter_contract_written', False): + return + rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 + parameters = [] + for chunk_index, model in enumerate(self.unwrapped_models): + for name, param in model.named_parameters(): + parameters.append({ + 'chunk': chunk_index, + 'name': name, + **self._parameter_record(param), + }) + payload = { + 'schema': 'glm52-loaded-parameter-inventory/v1', + 'framework': 'torch', + 'rank': rank, + 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, + 'parameters': parameters, + 'parameter_count': len(parameters), + 'local_numel': sum(item['numel'] for item in parameters), + } + os.makedirs(output_dir, exist_ok=True) + path = os.path.join(output_dir, f'rank{rank}.json') + with open(path, 'w', encoding='utf-8') as stream: + json.dump(payload, stream, ensure_ascii=False, indent=2, sort_keys=True) + stream.write('\n') + self._parameter_contract_written = True + + def _write_input_contract_once(self, data, seq_lens=None): + self._write_parameter_contract_once() + self._install_forward_contract_once() + path = os.environ.get('MODEL_REPRO_INPUT_RECEIPT_PATH') + if not path or getattr(self, '_input_contract_written', False): + return + if not mpu.is_pipeline_last_stage(ignore_virtual=False): + return + if torch.distributed.is_initialized() and torch.distributed.get_rank() != torch.distributed.get_world_size() - 1: + return + input_ids = data.get('input_ids') + labels = data.get('labels') + if input_ids is None or labels is None: + return + + def values(tensor): + return tensor.detach().to(device='cpu', dtype=torch.int64).reshape(-1).tolist() + + def digest(items): + return hashlib.sha256(json.dumps(items, separators=(',', ':')).encode()).hexdigest() + + input_values = values(input_ids) + label_values = values(labels) + model_mask_values = [label != -100 for label in label_values] + semantic_length = seq_lens[0] if seq_lens else len(input_values) + labels_were_shifted = self.args.task_type == 'causal_lm' and not getattr( + self.args, 'pretokenized_dataset', False) + semantic_input_values, semantic_label_values, semantic_mask_values = project_owning_loader_semantics( + input_values, label_values, semantic_length, labels_were_shifted) + payload = { + 'schema': 'glm52-owning-loader-input/v1', + 'framework': 'torch', + 'rank': torch.distributed.get_rank() if torch.distributed.is_initialized() else 0, + 'step': self.state.iteration + 1, + 'input_ids': { + 'shape': list(input_ids.shape), + 'dtype': str(input_ids.dtype), + 'count': len(input_values), + 'sha256': digest(input_values), + }, + 'labels': { + 'shape': list(labels.shape), + 'dtype': str(labels.dtype), + 'count': len(label_values), + 'supervised_count': sum(model_mask_values), + 'sha256': digest(label_values), + 'projection': 'model_next_token_labels', + }, + 'loss_mask': { + 'shape': list(labels.shape), + 'dtype': 'bool', + 'count': len(model_mask_values), + 'supervised_count': sum(model_mask_values), + 'sha256': digest(model_mask_values), + }, + 'semantic': { + 'input_token_count': len(semantic_input_values), + 'supervised_target_count': sum(semantic_mask_values), + 'input_ids_sha256': digest(semantic_input_values), + 'labels_sha256': digest(semantic_label_values), + 'loss_mask_sha256': digest(semantic_mask_values), + 'projection': 'dataset_row_before_megatron_padding_and_label_roll', + }, + 'carrier_padding': { + 'count': len(input_values) - len(semantic_input_values), + 'input_ids_sha256': digest(input_values[len(semantic_input_values):]), + 'labels_sha256': digest(label_values[len(semantic_input_values):]), + }, + 'ignore_index': -100, + 'dataset': os.environ.get('MODEL_REPRO_INPUT_DATASET_PATH'), + } + path = os.path.abspath(os.path.expanduser(path)) + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, 'w', encoding='utf-8') as stream: + json.dump(payload, stream, ensure_ascii=False, indent=2, sort_keys=True) + stream.write('\n') + self._input_contract_written = True + def seq_cls_loss_func(self, output_tensor, *, labels: torch.Tensor, packed_seq_params=None, attention_mask=None): args = self.args logits = self.get_last_tokens(output_tensor, packed_seq_params, attention_mask) @@ -120,6 +433,8 @@ def _compute_channel_loss(self, losses, loss_mask, channels, packed_seq_params=N def forward_step(self, data_iterator, model): vp_stage = model.module.module.vp_stage data = self.get_batch(data_iterator, vp_stage) + seq_lens = data.pop('_model_repro_seq_lens', None) + self._write_input_contract_once(data, seq_lens) loss_scale = data.pop('loss_scale', None) channels = data.pop('channel', None) labels = data.get('labels') diff --git a/swift/megatron/trainers/utils.py b/swift/megatron/trainers/utils.py index 16ca70d9a2..1817fca687 100644 --- a/swift/megatron/trainers/utils.py +++ b/swift/megatron/trainers/utils.py @@ -1,5 +1,6 @@ # Copyright (c) ModelScope Contributors. All rights reserved. import gc +import os import torch from accelerate.utils import gather as hf_gather from accelerate.utils import gather_object as hf_gather_object @@ -18,7 +19,7 @@ def get_batch_on_this_pp_rank(args, data, vp_stage=None): - if args.task_type == 'causal_lm': + if args.task_type == 'causal_lm' and not getattr(args, 'pretokenized_dataset', False): data['labels'] = torch.roll(data['labels'], -1, dims=-1) if 'loss_scale' in data: data['loss_scale'] = torch.roll(data['loss_scale'], -1, dims=-1) @@ -400,6 +401,10 @@ def prepare_batch(args, data, vp_stage=None): if num_samples is not None: batch['packed_seq_params'].num_samples = num_samples batch = get_batch_on_this_cp_rank(args, batch) + if os.environ.get('MODEL_REPRO_INPUT_RECEIPT_PATH') and seq_lens is not None: + # Opt-in metadata for the owning-loader receipt. MegatronTrainer removes + # it before model(**data), so it cannot alter the numerical path. + batch['_model_repro_seq_lens'] = [int(length) for length in seq_lens] return batch diff --git a/swift/model/register.py b/swift/model/register.py index e106929d13..842b94e5f6 100644 --- a/swift/model/register.py +++ b/swift/model/register.py @@ -174,6 +174,7 @@ def __init__( auto_model_cls=None, return_dummy_model: bool = False, new_special_tokens: Optional[List[str]] = None, + processor_id_or_path: Optional[str] = None, model_kwargs: Optional[Dict[str, Any]] = None, **kwargs, ): @@ -197,6 +198,7 @@ def __init__( self.auto_tokenizer_cls = None self.return_dummy_model = return_dummy_model self.new_special_tokens = new_special_tokens + self.processor_id_or_path = processor_id_or_path self.model_kwargs = model_kwargs self.patch_offload = kwargs.pop('patch_offload', False) self.init_strategy = kwargs.get('init_strategy') @@ -257,15 +259,16 @@ def _get_tokenizer(self, processor): return tokenizer def get_processor(self, model_dir: str, config: PretrainedConfig) -> Processor: + processor_dir = self.processor_id_or_path or model_dir auto_tokenizer_cls = self.auto_tokenizer_cls if auto_tokenizer_cls is None: - if os.path.exists(os.path.join(model_dir, 'preprocessor_config.json')) or os.path.exists( - os.path.join(model_dir, 'processor_config.json')): + if os.path.exists(os.path.join(processor_dir, 'preprocessor_config.json')) or os.path.exists( + os.path.join(processor_dir, 'processor_config.json')): from transformers import AutoProcessor auto_tokenizer_cls = AutoProcessor else: auto_tokenizer_cls = AutoTokenizer - return auto_tokenizer_cls.from_pretrained(model_dir, trust_remote_code=self.default_trust_remote_code) + return auto_tokenizer_cls.from_pretrained(processor_dir, trust_remote_code=self.default_trust_remote_code) def get_model(self, model_dir: str, config: PretrainedConfig, processor: Processor, model_kwargs) -> PreTrainedModel: @@ -534,6 +537,7 @@ def get_model_processor( max_model_len: Optional[int] = None, auto_model_cls=None, new_special_tokens: Optional[List[str]] = None, + processor_id_or_path: Optional[str] = None, task_type: Literal['causal_lm', 'seq_cls', 'embedding', 'reranker', 'generative_reranker'] = None, num_labels: Optional[int] = None, problem_type: Literal['regression', 'single_label_classification', 'multi_label_classification'] = None, @@ -568,6 +572,7 @@ def get_model_processor( max_model_len: Maximum sequence length the model can handle. auto_model_cls: Custom AutoModel class to use for loading (e.g., AutoModelForCausalLM). new_special_tokens: List of new special tokens to add to the tokenizer. + processor_id_or_path: Optional independent tokenizer or processor source. task_type: Task type for the model. Options: 'causal_lm', 'seq_cls', 'embedding', 'reranker', 'generative_reranker'. num_labels: Number of labels for classification tasks. @@ -625,6 +630,7 @@ def get_model_processor( auto_model_cls=auto_model_cls, return_dummy_model=return_dummy_model, new_special_tokens=new_special_tokens, + processor_id_or_path=processor_id_or_path, model_kwargs=model_kwargs, **kwargs) return loader.load() diff --git a/swift/pipelines/train/sft.py b/swift/pipelines/train/sft.py index 8d547f4106..13eb0f3bcf 100644 --- a/swift/pipelines/train/sft.py +++ b/swift/pipelines/train/sft.py @@ -5,7 +5,7 @@ from swift.arguments import SftArguments from swift.dataset import (AddLengthPreprocessor, DatasetLoader, EncodePreprocessor, IterablePackingDataset, - LazyLLMDataset, PackingDataset) + LazyLLMDataset, PackingDataset, validate_pretokenized_dataset) from swift.infer_engine import prepare_generation_config from swift.ray_utils import RayHelper from swift.sequence_parallel import sequence_parallel @@ -49,6 +49,8 @@ def _prepare_generation_config(self): def _prepare_model_tokenizer(self, **kwargs): args = self.args self.model, self.processor = args.get_model_processor(**kwargs) + if getattr(args, 'tokenizer_name_or_path', None): + logger.info(f'Using independent tokenizer_name_or_path: {args.tokenizer_name_or_path}') if args.sequence_parallel_size > 1: sequence_parallel.prepare( args.sequence_parallel_size, model=self.model, tokenizer=self.processor, padding_free=args.padding_free) @@ -124,6 +126,12 @@ def _prepare_dataset(self): def _post_process_datasets(self, datasets: List) -> List: args = self.args + if args.pretokenized_dataset: + for dataset in datasets: + if dataset is not None: + validate_pretokenized_dataset(dataset, args.max_length) + logger.info('Using pretokenized cached dataset without template encoding.') + return datasets predict_with_generate = getattr(args, 'predict_with_generate', False) template = self.template diff --git a/tests/general/test_dataset_empty_assistant.py b/tests/general/test_dataset_empty_assistant.py new file mode 100644 index 0000000000..cff427ceba --- /dev/null +++ b/tests/general/test_dataset_empty_assistant.py @@ -0,0 +1,31 @@ +from datasets import Dataset + +from swift.dataset.loader import DatasetLoader + + +def test_empty_assistant_filter_is_default_off(): + dataset = Dataset.from_list( + [ + {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": ""}]}, + {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": "a"}]}, + ] + ) + + filtered = DatasetLoader()._filter_empty_assistant_responses(dataset) + + assert len(filtered) == 2 + + +def test_empty_assistant_filter_drops_only_blank_assistant_content(): + dataset = Dataset.from_list( + [ + {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": ""}]}, + {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": " "}]}, + {"messages": [{"role": "user", "content": ""}, {"role": "assistant", "content": "a"}]}, + ] + ) + + filtered = DatasetLoader(drop_empty_assistant_response=True)._filter_empty_assistant_responses(dataset) + + assert len(filtered) == 1 + assert filtered[0]["messages"][-1]["content"] == "a" diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 70a5c579bd..d3c10f0ac8 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -2,6 +2,7 @@ from transformers import PretrainedConfig from types import SimpleNamespace +from swift.megatron.init import _get_save_processor_id from swift.megatron.model import utils @@ -37,6 +38,15 @@ def _patch_model_config(monkeypatch): SimpleNamespace(name='num_moe_experts')]) +def test_save_processor_prefers_independent_tokenizer_source(): + assert _get_save_processor_id( + SimpleNamespace(model_dir='/weights', tokenizer_name_or_path='/tokenizer') + ) == '/tokenizer' + assert _get_save_processor_id( + SimpleNamespace(model_dir='/weights', tokenizer_name_or_path=None) + ) == '/weights' + + def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) @@ -71,3 +81,28 @@ def test_get_mcore_model_config_keeps_explicit_mtp_num_layers(monkeypatch): config = utils.get_mcore_model_config(_make_args(mtp_num_layers=2), hf_config) assert config.kwargs['mtp_num_layers'] == 2 + + +def test_dsa_index_share_allows_recompute_none(): + config = SimpleNamespace( + experimental_attention_variant='dsa', + dsa_indexer_topk_freq=4, + recompute_granularity='none', + ) + + utils._check_dsa_index_share_recompute(config) + + +def test_dsa_index_share_rejects_selective_recompute(): + config = SimpleNamespace( + experimental_attention_variant='dsa', + dsa_indexer_topk_freq=4, + recompute_granularity='selective', + ) + + try: + utils._check_dsa_index_share_recompute(config) + except ValueError as error: + assert 'Set recompute_granularity=none' in str(error) + else: + raise AssertionError('expected DSA index sharing with selective recompute to fail closed') diff --git a/tests/megatron/test_pretokenized_dataset.py b/tests/megatron/test_pretokenized_dataset.py new file mode 100644 index 0000000000..523131d8d5 --- /dev/null +++ b/tests/megatron/test_pretokenized_dataset.py @@ -0,0 +1,66 @@ +from types import SimpleNamespace + +import pytest +import torch +from datasets import Dataset + +from swift.dataset import validate_pretokenized_dataset +from swift.megatron.trainers import utils as trainer_utils +from swift.model.register import ModelLoader + + +def test_validate_pretokenized_dataset_accepts_fixed_tokens(): + dataset = Dataset.from_dict({ + 'input_ids': [[154820, 42, 42, 17, 99, 42, 8]], + 'labels': [[42, 42, 17, 99, 42, 8, 3]], + 'position_ids': [[0, 1, 2, 3, 4, 5, 6]], + 'lengths': [7], + }) + + validate_pretokenized_dataset(dataset, max_length=7) + + +def test_validate_pretokenized_dataset_rejects_missing_labels(): + dataset = Dataset.from_dict({ + 'input_ids': [[1, 2]], + 'position_ids': [[0, 1]], + 'lengths': [2], + }) + + with pytest.raises(ValueError, match='missing columns'): + validate_pretokenized_dataset(dataset, max_length=2) + + +def test_pretokenized_labels_are_not_shifted(monkeypatch): + labels = torch.tensor([[42, 42, 17, 99, 42, 8, 3]]) + data = {'input_ids': torch.zeros_like(labels), 'labels': labels.clone()} + args = SimpleNamespace( + task_type='causal_lm', + pretokenized_dataset=True, + pipeline_model_parallel_size=1, + ) + monkeypatch.setattr(trainer_utils, 'get_current_device', lambda: 'cpu') + monkeypatch.setattr(trainer_utils, 'to_device', lambda value, *_args, **_kwargs: value) + + batch = trainer_utils.get_batch_on_this_pp_rank(args, data) + + assert torch.equal(batch['labels'], labels) + + +def test_model_loader_uses_independent_processor_path(): + requested = [] + + class FakeTokenizer: + @classmethod + def from_pretrained(cls, path, **kwargs): + requested.append((path, kwargs)) + return object() + + loader = ModelLoader.__new__(ModelLoader) + loader.processor_id_or_path = '/tokenizer-only' + loader.auto_tokenizer_cls = FakeTokenizer + loader.default_trust_remote_code = True + + loader.get_processor('/weights-only', config=None) + + assert requested == [('/tokenizer-only', {'trust_remote_code': True})] diff --git a/tests/megatron/test_raw_loss_observability.py b/tests/megatron/test_raw_loss_observability.py new file mode 100644 index 0000000000..58c5f6a1f2 --- /dev/null +++ b/tests/megatron/test_raw_loss_observability.py @@ -0,0 +1,72 @@ +import torch + +from swift.megatron.callbacks.print import raw_loss_event +from swift.megatron.trainers.trainer import MegatronTrainer, project_owning_loader_semantics + + +def test_raw_loss_event_preserves_unrounded_values_and_step(): + logs = { + "loss": 12.410510059999999, + "mtp_0_loss": 13.367947579999999, + "eval_loss": 99.0, + } + assert raw_loss_event(1, logs) == { + "step": 1, + "loss": 12.410510059999999, + "mtp_0_loss": 13.367947579999999, + } + + +def test_raw_loss_event_omits_non_loss_metrics(): + assert raw_loss_event(3, {"grad_norm": 1.0, "learning_rate": 1e-6}) is None + + +def test_owning_loader_projection_removes_sp_padding_and_reverses_label_roll(): + input_values = list(range(57)) + [154820] + original_labels = [-100] * 13 + list(range(44)) + [-100] + shifted_labels = original_labels[1:] + original_labels[:1] + semantic_input, semantic_labels, semantic_mask = project_owning_loader_semantics( + input_values, shifted_labels, semantic_length=57, labels_were_shifted=True) + assert semantic_input == list(range(57)) + assert semantic_labels == original_labels[:57] + assert sum(semantic_mask) == 44 + + +def test_owning_loader_projection_rejects_invalid_semantic_length(): + try: + project_owning_loader_semantics([1, 2], [-100, 2], semantic_length=3) + except ValueError as exc: + assert "invalid owning-loader semantic length" in str(exc) + else: + raise AssertionError("invalid semantic length was accepted") + + +def test_parameter_record_preserves_orientation_and_signed_zero(): + tensor = torch.tensor([[0.0, -0.0], [1.0, 2.0]], dtype=torch.float32) + record = MegatronTrainer._parameter_record(tensor) + assert record["shape"] == [2, 2] + assert record["dtype"] == "torch.float32" + assert record["positive_zero_count"] == 1 + assert record["negative_zero_count"] == 1 + assert record["sha256"] != record["transpose_sha256"] + + +def test_layer0_fine_forward_specs_are_explicit_and_fail_closed(): + assert MegatronTrainer._forward_contract_specs("coarse") is None + specs = MegatronTrainer._forward_contract_specs("layer0_fine") + assert len(specs) == 13 + assert specs["decoder.layers.0.input_layernorm"] == "layer0_input_rmsnorm_output" + assert specs["decoder.layers.0.self_attention.linear_q_down_proj"] == "layer0_q_down_projection_output" + assert specs["decoder.layers.0"] == "base_transformer_layer_0_output" + try: + MegatronTrainer._forward_contract_specs("unknown") + except ValueError as exc: + assert "unsupported MODEL_REPRO_FORWARD_BOUNDARY_SET" in str(exc) + else: + raise AssertionError("unknown forward boundary set was accepted") + + +def test_first_tensor_prefers_first_tensor_in_nested_module_output(): + first = torch.tensor([1.0]) + second = torch.tensor([2.0]) + assert MegatronTrainer._first_tensor((None, {"output": first}, second)) is first From cb1de4d5517a5e74e7141b31aa4d2152dd6e02b6 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 27 Aug 2026 17:01:42 +0800 Subject: [PATCH 06/33] Force LocalSpecProvider for DSA when accuracy-compatible is on _replace_spec_dsa otherwise still asks TESpecProvider, leaving TELinear on indexer/MLA while Paddle HAVE_TE is False. --- swift/megatron/init.py | 12 ++++++++++++ tests/megatron/test_model_config.py | 14 ++++++++++++++ 2 files changed, 26 insertions(+) diff --git a/swift/megatron/init.py b/swift/megatron/init.py index ef294657a9..9209de3e2b 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -208,6 +208,18 @@ def wrapper(*args, **kwargs): mcb_register.get_gpt_decoder_block_spec = _force_local_spec(mcb_register.get_gpt_decoder_block_spec) mcb_register.get_gpt_mtp_block_spec = _force_local_spec(mcb_register.get_gpt_mtp_block_spec) + # 1b) DSA is swapped in after the decoder spec via ModelLoader._replace_spec_dsa, + # which called _get_backend_spec_provider (TESpecProvider). That left TELinear / + # TENorm on DSA indexer + MLA while PaddleFleet HAVE_TE is False. E-259: remaining + # Torch DSA TE contaminated post_attn_norm. Force LocalSpecProvider instead. + from megatron.core.models.gpt import experimental_attention_variant_module_specs as _eav + + def _local_backend_spec_provider(config): + from megatron.core.models.backends import LocalSpecProvider + return LocalSpecProvider() + + _eav._get_backend_spec_provider = _local_backend_spec_provider + # 2) persist_layer_norm=False on the model config (dataclass default is baked into # __init__, so flip it on the instance via __post_init__). from mcore_bridge.config.model_config import ModelConfig as McbModelConfig diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index d3c10f0ac8..3b7aa8d732 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -93,6 +93,20 @@ def test_dsa_index_share_allows_recompute_none(): utils._check_dsa_index_share_recompute(config) +def test_dsa_backend_forced_to_local_spec_when_accuracy_compatible(monkeypatch): + from megatron.core.models.backends import LocalSpecProvider + from megatron.core.models.gpt import experimental_attention_variant_module_specs as eav + + import swift.megatron.init as init + + monkeypatch.setattr(init, '_use_accuracy_compatible_enabled', lambda: True) + init._patch_mcore_bridge_disable_te() + provider = eav._get_backend_spec_provider(SimpleNamespace()) + assert isinstance(provider, LocalSpecProvider) + assert hasattr(provider, 'linear') + assert provider.linear() is not provider.column_parallel_linear() + + def test_dsa_index_share_rejects_selective_recompute(): config = SimpleNamespace( experimental_attention_variant='dsa', From 691711f737ca5132fe7c0e482e565f1b181f311f Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Fri, 4 Sep 2026 12:32:26 +0800 Subject: [PATCH 07/33] fix: accept deterministic_mode and num_nextn_predict_layers CLI Stack-top dump-off YAML uses those keys. Without the dataclass fields parse_args left remaining_argv and failed closed. Alias MTP depth onto mtp_num_layers and apply the Megatron deterministic contract at init. --- swift/megatron/arguments/megatron_args.py | 4 ++++ swift/megatron/utils/megatron_lm_utils.py | 25 +++++++++++++++++++++++ 2 files changed, 29 insertions(+) diff --git a/swift/megatron/arguments/megatron_args.py b/swift/megatron/arguments/megatron_args.py index bc60b9d7a3..2af6fdb0c7 100644 --- a/swift/megatron/arguments/megatron_args.py +++ b/swift/megatron/arguments/megatron_args.py @@ -609,6 +609,7 @@ class MegatronArguments(RLHFMegatronArgumentsMixin, MegatronTunerMixin): overlap_param_gather: bool = False overlap_param_gather_with_optimizer_step: bool = False align_grad_reduce: bool = True + deterministic_mode: bool = False # Eagerly create NCCL communicators before the training loop to avoid the lazy # first-use allocation hitting the iteration-1 memory peak (Failed to CUDA calloc async). nccl_comm_warmup: bool = False @@ -668,6 +669,7 @@ class MegatronArguments(RLHFMegatronArgumentsMixin, MegatronTunerMixin): # mtp mtp_num_layers: Optional[int] = None + num_nextn_predict_layers: Optional[int] = None mtp_loss_scaling_factor: float = 0.1 mtp_decoder_input_detach: bool = False mtp_shared_weights: bool = False @@ -798,6 +800,8 @@ def __post_init__(self): logger.warning(f'Failed to sync dummy template suffix for use_accuracy_compatible: {e}') self._check_mcore_bridge() + if self.mtp_num_layers is None and self.num_nextn_predict_layers is not None: + self.mtp_num_layers = self.num_nextn_predict_layers if self.recompute_granularity == 'none': self.recompute_granularity = None diff --git a/swift/megatron/utils/megatron_lm_utils.py b/swift/megatron/utils/megatron_lm_utils.py index 55e4c1d80f..b8d4618189 100644 --- a/swift/megatron/utils/megatron_lm_utils.py +++ b/swift/megatron/utils/megatron_lm_utils.py @@ -83,7 +83,32 @@ def _initialize_mpu(args): f'EP: {args.expert_model_parallel_size}, ETP: {args.expert_tensor_parallel_size}') +def configure_deterministic_mode(args): + """Enable the same fail-closed deterministic contract as Megatron-LM training.""" + if not getattr(args, 'deterministic_mode', False): + return + attention_backend = getattr(args, 'attention_backend', '') + attention_backend = getattr(attention_backend, 'name', str(attention_backend)).lower() + if attention_backend == 'flash': + raise ValueError('Flash attention cannot be used in deterministic mode.') + if getattr(args, 'cross_entropy_loss_fusion', False): + raise ValueError('Cross entropy fusion cannot be used in deterministic mode.') + nccl_algo = os.environ.get('NCCL_ALGO') + allowed_nccl_algorithms = {'Tree', 'Ring', 'CollnetDirect', 'CollnetChain', '^NVLS'} + if nccl_algo not in allowed_nccl_algorithms: + raise ValueError( + f'NCCL_ALGO must be explicitly set to one of {sorted(allowed_nccl_algorithms)} ' + 'in deterministic mode.' + ) + torch.use_deterministic_algorithms(True) + torch.backends.cudnn.deterministic = True + torch.backends.cudnn.benchmark = False + logger.info(f'Deterministic mode enabled with NCCL_ALGO={nccl_algo}.') + + def initialize_megatron(args): + configure_deterministic_mode(args) + # Pytorch distributed. _initialize_mpu(args) From 2f20c5a54948fd99cf8a6e374f33caa05d2f5311 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Fri, 4 Sep 2026 12:49:37 +0800 Subject: [PATCH 08/33] fix: map local-spec dense MLP norm to pre_mlp_layernorm TE-off local TransformerLayer has a standalone pre_mlp_layernorm, not fused linear_fc1.layer_norm_weight. Weight load asserted None on that fused key. Route the HF post-attention LN onto the local module when the fused parameter is absent. --- swift/megatron/init.py | 35 ++++++++++++++++++++++++++++- tests/megatron/test_model_config.py | 11 ++++++++- 2 files changed, 44 insertions(+), 2 deletions(-) diff --git a/swift/megatron/init.py b/swift/megatron/init.py index 9209de3e2b..db5a2d599a 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -254,7 +254,40 @@ def _set_layer_attn(self, mg_layer, hf_state_dict, layer_idx, to_mcore): return hf_state_dict McbGPTBridge._set_layer_attn = _set_layer_attn - logger.info('mcore_bridge patched for TE-off alignment (local spec, persist_layer_norm=False, input_layernorm map)') + + # 4) local-spec dense-MLP norm key mapping: with the local (non-TE) spec the + # dense MLP is `ColumnParallelLinear` + separate `pre_mlp_layernorm` + # (no fused `linear_fc1.layer_norm_weight`); route that key accordingly. + def _set_layer_mlp(self, mg_layer, hf_state_dict, layer_idx, to_mcore, is_mtp=False): + mg_mlp = None if mg_layer is None else mg_layer.mlp + is_moe = True if mg_mlp is not None and hasattr(mg_mlp, 'experts') else False + if not to_mcore: + is_moe = torch.tensor([is_moe], dtype=torch.bool, device='cuda') + if self.pp_size > 1: + dist.all_reduce(is_moe, group=self.pp_group) + if is_moe: + hf_state_dict.update( + self._set_moe_state( + mg_mlp, hf_state_dict, f'{self.hf_mlp_prefix}.', layer_idx, to_mcore, is_mtp=is_mtp)) + self._set_state_dict(mg_layer, 'pre_mlp_layernorm.weight', hf_state_dict, + self.hf_post_attention_layernorm_key, to_mcore) + else: + hf_state_dict.update( + self._set_mlp_state(mg_mlp, hf_state_dict, f'{self.hf_mlp_prefix}.', layer_idx, to_mcore)) + mg_fc1 = None if mg_layer is None else getattr(getattr(mg_layer, 'mlp', None), 'linear_fc1', None) + fused_norm_weight = getattr(mg_fc1, 'layer_norm_weight', None) + if fused_norm_weight is None: + self._set_state_dict(mg_layer, 'pre_mlp_layernorm.weight', hf_state_dict, + self.hf_post_attention_layernorm_key, to_mcore) + else: + self._set_state_dict(mg_layer, 'mlp.linear_fc1.layer_norm_weight', hf_state_dict, + self.hf_post_attention_layernorm_key, to_mcore) + return hf_state_dict + + McbGPTBridge._set_layer_mlp = _set_layer_mlp + logger.info( + 'mcore_bridge patched for TE-off alignment (local spec, persist_layer_norm=False, input_layernorm+mlp-norm map)' + ) def _patch_mcore_bridge(): diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 3b7aa8d732..f48d159607 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -1,8 +1,10 @@ +import inspect + import torch from transformers import PretrainedConfig from types import SimpleNamespace -from swift.megatron.init import _get_save_processor_id +from swift.megatron.init import _get_save_processor_id, _patch_mcore_bridge_disable_te from swift.megatron.model import utils @@ -107,6 +109,13 @@ def test_dsa_backend_forced_to_local_spec_when_accuracy_compatible(monkeypatch): assert provider.linear() is not provider.column_parallel_linear() +def test_local_spec_mlp_norm_maps_pre_mlp_layernorm_when_unfused(): + source = inspect.getsource(_patch_mcore_bridge_disable_te) + assert "fused_norm_weight is None" in source + assert "pre_mlp_layernorm.weight" in source + assert "mlp.linear_fc1.layer_norm_weight" in source + + def test_dsa_index_share_rejects_selective_recompute(): config = SimpleNamespace( experimental_attention_variant='dsa', From f81acbc9436c8acd112178a89cd6b9a2c6d2c632 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 10:38:16 +0800 Subject: [PATCH 09/33] fix: pad sequence-parallel collate to TP*2 TP2+SP pads ceil(57/2)*2=58 without the extra *2, while PaddleFleet and the E-811 IEEE 1-100 carrier are 60. Restore the SP interleave factor so get_padding_to is 4 on the frozen profile. Signed-off-by: Zhan Rongrui --- swift/megatron/utils/utils.py | 4 ++++ tests/megatron/test_model_config.py | 20 ++++++++++++++++++++ 2 files changed, 24 insertions(+) diff --git a/swift/megatron/utils/utils.py b/swift/megatron/utils/utils.py index 8ba40d44ab..44462ecebb 100644 --- a/swift/megatron/utils/utils.py +++ b/swift/megatron/utils/utils.py @@ -210,6 +210,10 @@ def get_padding_to(args): padding_to = None if args.tensor_model_parallel_size > 1 and args.sequence_parallel: padding_to = args.tensor_model_parallel_size + # Sequence-parallel interleaves the sequence across TP ranks at 2x. + # Without this the collator pads to ceil(57/TP)*TP=58 while the + # PaddleFleet carrier and E-811 IEEE 1-100 are 60 (57 -> TP*SP=4). + padding_to = padding_to * 2 if args.context_parallel_size > 1: padding_to = (padding_to or 1) * args.context_parallel_size origin_padding_to = padding_to diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index f48d159607..f380f265ce 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -6,6 +6,7 @@ from swift.megatron.init import _get_save_processor_id, _patch_mcore_bridge_disable_te from swift.megatron.model import utils +from swift.megatron.utils.utils import get_padding_to class _ModelConfigStub: @@ -85,6 +86,25 @@ def test_get_mcore_model_config_keeps_explicit_mtp_num_layers(monkeypatch): assert config.kwargs['mtp_num_layers'] == 2 +def test_get_padding_to_sequence_parallel_uses_tp_times_two(): + args = SimpleNamespace( + tensor_model_parallel_size=2, + sequence_parallel=True, + context_parallel_size=1, + fp8_recipe='delayed', + fp8_format=None, + fp8=None, + fp4_format=None, + fp4=None, + attention_backend='unfused', + ) + assert get_padding_to(args) == 4 + seq_len = 57 + import math + assert math.ceil(seq_len / 4) * 4 == 60 + assert math.ceil(seq_len / 2) * 2 == 58 + + def test_dsa_index_share_allows_recompute_none(): config = SimpleNamespace( experimental_attention_variant='dsa', From a735ae06440db1eb1ef60da6bc48d07580c443c6 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 10:44:59 +0800 Subject: [PATCH 10/33] style: isort/yapf the sequence-parallel padding_to test CI flake8/isort/yapf on PFCCLab/ms-swift#3 failed on the new test imports and nearby wrap. Numerics unchanged. Signed-off-by: Zhan Rongrui --- tests/megatron/test_model_config.py | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index f380f265ce..2245461930 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -1,5 +1,5 @@ import inspect - +import math import torch from transformers import PretrainedConfig from types import SimpleNamespace @@ -42,12 +42,9 @@ def _patch_model_config(monkeypatch): def test_save_processor_prefers_independent_tokenizer_source(): - assert _get_save_processor_id( - SimpleNamespace(model_dir='/weights', tokenizer_name_or_path='/tokenizer') - ) == '/tokenizer' - assert _get_save_processor_id( - SimpleNamespace(model_dir='/weights', tokenizer_name_or_path=None) - ) == '/weights' + assert _get_save_processor_id(SimpleNamespace(model_dir='/weights', + tokenizer_name_or_path='/tokenizer')) == '/tokenizer' + assert _get_save_processor_id(SimpleNamespace(model_dir='/weights', tokenizer_name_or_path=None)) == '/weights' def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): @@ -100,7 +97,6 @@ def test_get_padding_to_sequence_parallel_uses_tp_times_two(): ) assert get_padding_to(args) == 4 seq_len = 57 - import math assert math.ceil(seq_len / 4) * 4 == 60 assert math.ceil(seq_len / 2) * 2 == 58 From 65b897a6e92470dee459f541ebde0605d6a118c9 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 11:19:33 +0800 Subject: [PATCH 11/33] style: apply pre-commit after merging main CI files Merge PFCCLab/ms-swift main so lint --all-files includes the new alignment workflow and build.sh. Wrap the two E501 lines in the observability trainer helpers and let isort/yapf/single-quote hooks rewrite only files already on this branch. --- .../workflows/alignment_model_accuracy.yaml | 4 +- scripts/dependence/build.sh | 1 - swift/megatron/callbacks/print.py | 3 +- swift/megatron/init.py | 108 +++++++----------- swift/megatron/model/utils.py | 10 +- swift/megatron/trainers/trainer.py | 61 ++++++---- swift/megatron/utils/megatron_lm_utils.py | 6 +- tests/general/test_dataset_empty_assistant.py | 64 ++++++++--- tests/megatron/test_model_config.py | 6 +- tests/megatron/test_pretokenized_dataset.py | 4 +- tests/megatron/test_raw_loss_observability.py | 46 ++++---- 11 files changed, 162 insertions(+), 151 deletions(-) diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index 76faf444ba..6de447d2e4 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -65,7 +65,7 @@ jobs: wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar cd PaddleFleet && git pull && cd - - + echo "Download ms-swift form https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz" wget -q --no-proxy https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate rm -rf ms-swift && tar zxf ms-swift.tar.gz && rm -rf ms-swift.tar.gz @@ -145,7 +145,7 @@ jobs: export MEGATRON_CORE_WHEEL_PATH=/workspace/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl export MS_SWIFT_WHEEL_PATH=/workspace/upload/ms_swift-0.0.0-py3-none-any.whl export MCORE_BRIDGE_WHEEL_PATH=/workspace/mcore_bridge-0.0.0-py3-none-any.whl - # The BOS wheels are republished under a fixed 0.0.0 filename + # The BOS wheels are republished under a fixed 0.0.0 filename export UV_SKIP_WHEEL_FILENAME_CHECK=1 for whl in "$PADDLEFLEET_WHEEL_PATH" "$PADDLEFLEET_OPS_WHEEL_PATH" \ diff --git a/scripts/dependence/build.sh b/scripts/dependence/build.sh index de73702692..6b9129513c 100644 --- a/scripts/dependence/build.sh +++ b/scripts/dependence/build.sh @@ -79,4 +79,3 @@ echo -e "\033[32m ---- make ms-swift.tar.gz \033[0m" swift_tar echo -e "\033[32m ---- build ms-swift whl \033[0m" swift_build - diff --git a/swift/megatron/callbacks/print.py b/swift/megatron/callbacks/print.py index 83ec6f71c5..4b8a897f68 100644 --- a/swift/megatron/callbacks/print.py +++ b/swift/megatron/callbacks/print.py @@ -15,8 +15,7 @@ def raw_loss_event(step, logs): """Return an unrounded training-loss event, excluding evaluation metrics.""" raw_losses = { key: value - for key, value in logs.items() - if key == 'loss' or (key.startswith('mtp_') and key.endswith('_loss')) + for key, value in logs.items() if key == 'loss' or (key.startswith('mtp_') and key.endswith('_loss')) } return {'step': step, **raw_losses} if raw_losses else None diff --git a/swift/megatron/init.py b/swift/megatron/init.py index db5a2d599a..943387db7a 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -16,22 +16,15 @@ from typing import Optional from swift.model import get_model_processor, save_checkpoint -from swift.utils import ( - HfConfigFactory, - disable_safe_ddp_context_use_barrier, - get_logger, - get_modules_to_not_convert, - get_multimodal_target_regex, - is_master, - split_list, -) +from swift.utils import (HfConfigFactory, disable_safe_ddp_context_use_barrier, get_logger, get_modules_to_not_convert, + get_multimodal_target_regex, is_master, split_list) logger = get_logger() def _get_save_processor_id(args): """Use the configured processor source when weights and tokenizer are independent.""" - return getattr(args, "tokenizer_name_or_path", None) or args.model_dir + return getattr(args, 'tokenizer_name_or_path', None) or args.model_dir def _patch__batched_p2p_ops(): @@ -40,7 +33,7 @@ def _patch__batched_p2p_ops(): _batched_p2p_ops_origin = p2p_communication._batched_p2p_ops def _batched_p2p_ops(**kwargs): - kwargs["group"] = None + kwargs['group'] = None return _batched_p2p_ops_origin(**kwargs) p2p_communication._batched_p2p_ops = _batched_p2p_ops @@ -52,7 +45,7 @@ def _patch_torch_FileSystemReader(): _origin_read_data = FileSystemReader.read_data _origin__slice_file = FileSystemReader._slice_file - READER_MAX_WORKERS = int(os.environ.get("MCORE_READER_MAX_WORKERS", "16")) + READER_MAX_WORKERS = int(os.environ.get('MCORE_READER_MAX_WORKERS', '16')) @contextmanager def _patch__slice_file(prog_bar): @@ -72,12 +65,10 @@ def read_data(self, plan, planner): def _worker(plan_shard): _origin_read_data(self, plan_shard, planner) - prog_bar = tqdm(total=len(plan.items), dynamic_ncols=True, desc="Loading: ") + prog_bar = tqdm(total=len(plan.items), dynamic_ncols=True, desc='Loading: ') plan_shards = split_list(plan.items, READER_MAX_WORKERS, contiguous=False) with _patch__slice_file(prog_bar): - with concurrent.futures.ThreadPoolExecutor( - max_workers=READER_MAX_WORKERS - ) as pool: + with concurrent.futures.ThreadPoolExecutor(max_workers=READER_MAX_WORKERS) as pool: futures = [] for i in range(READER_MAX_WORKERS): plan_shard = copy(plan) @@ -101,12 +92,8 @@ def _patch_validate_non_overlapping_shards_metadata(): def validate_non_overlapping_shards_metadata(*args, **kwargs): pass - api.validate_non_overlapping_shards_metadata = ( - validate_non_overlapping_shards_metadata - ) - api2.validate_non_overlapping_shards_metadata = ( - validate_non_overlapping_shards_metadata - ) + api.validate_non_overlapping_shards_metadata = (validate_non_overlapping_shards_metadata) + api2.validate_non_overlapping_shards_metadata = (validate_non_overlapping_shards_metadata) def _validate_global_plan(*args, **kwargs): # torch returns a list of error messages here; empty list means "no error". @@ -142,8 +129,8 @@ def _patch_unified_memory(): load_inline = cpp_extension.load_inline def _new_load_inline(*args, **kwargs): - name = kwargs.get("name") - if name == "managed_alloc_runtime": + name = kwargs.get('name') + if name == 'managed_alloc_runtime': raise RuntimeError return load_inline(*args, **kwargs) @@ -292,8 +279,8 @@ def _set_layer_mlp(self, mg_layer, hf_state_dict, layer_idx, to_mcore, is_mtp=Fa def _patch_mcore_bridge(): require_version( - "mcore-bridge>=1.4.0", - "please install mcore-bridge via `pip install mcore-bridge -U`", + 'mcore-bridge>=1.4.0', + 'please install mcore-bridge via `pip install mcore-bridge -U`', ) import mcore_bridge from mcore_bridge import GPTBridge @@ -301,12 +288,12 @@ def _patch_mcore_bridge(): logger.info(f"mcore_bridge.__version__: {mcore_bridge.__version__}") if _use_accuracy_compatible_enabled(): _patch_mcore_bridge_disable_te() - if not getattr(ModelLoader._replace_spec_dsa, "_swift_norm_accuracy_patch", False): + if not getattr(ModelLoader._replace_spec_dsa, '_swift_norm_accuracy_patch', False): origin_replace_spec_dsa = ModelLoader._replace_spec_dsa def replace_spec_dsa(self, layer_spec): origin_replace_spec_dsa(self, layer_spec) - if not getattr(self.config, "norm_accuracy_compatible", False): + if not getattr(self.config, 'norm_accuracy_compatible', False): return from megatron.core.transformer.torch_norm import WrappedTorchNorm @@ -323,7 +310,7 @@ def save_weights( mg_models, output_dir: str, peft_format: bool = False, - max_shard_size: str = "5GB", + max_shard_size: str = '5GB', args=None, processor=None, ) -> None: @@ -338,13 +325,13 @@ def save_weights( return hf_config = self.config.hf_config hf_config = deepcopy(hf_config) - if is_master() and not hasattr(self, "hf_model"): - if hasattr(self, "get_hf_meta_model"): + if is_master() and not hasattr(self, 'hf_model'): + if hasattr(self, 'get_hf_meta_model'): self.hf_model = self.get_hf_meta_model() self.hf_model.model_meta = processor.model_meta self.hf_model.model_info = processor.model_info else: - with torch.device("meta"), disable_safe_ddp_context_use_barrier(): + with torch.device('meta'), disable_safe_ddp_context_use_barrier(): self.hf_model = get_model_processor( args.model_dir, model_type=args.model_type, @@ -355,21 +342,20 @@ def save_weights( if is_master(): if peft_format: peft_config = copy(mg_models[0].peft_config[self._adapter_name]) - if self.config.task_type == "seq_cls": - peft_config.task_type = "SEQ_CLS" - if self.is_multimodal and "all-linear" in args.target_modules: + if self.config.task_type == 'seq_cls': + peft_config.task_type = 'SEQ_CLS' + if self.is_multimodal and 'all-linear' in args.target_modules: peft_config.target_modules = get_multimodal_target_regex( self.hf_model, freeze_llm=args.freeze_llm, freeze_vit=args.freeze_vit, freeze_aligner=args.freeze_aligner, - include_embedding="all-embedding" in args.target_modules, - exclude_router="all-router" not in args.target_modules, + include_embedding='all-embedding' in args.target_modules, + exclude_router='all-router' not in args.target_modules, ) else: assert not isinstance(peft_config.target_modules, str), ( - "target_regex is not currently supported for LoRA conversion. Please set `--merge_lora true`." - ) + 'target_regex is not currently supported for LoRA conversion. Please set `--merge_lora true`.') peft_config.target_modules = self._peft_target_modules peft_config.modules_to_save = self._peft_modules_to_save peft_config.save_pretrained(output_dir) @@ -377,38 +363,26 @@ def save_weights( config = self.config llm_config = HfConfigFactory.get_text_config(hf_config) if config.mtp_num_layers: - for key in ["num_nextn_predict_layers", "mtp_num_hidden_layers"]: + for key in ['num_nextn_predict_layers', 'mtp_num_hidden_layers']: if hasattr(llm_config, key): setattr(llm_config, key, config.mtp_num_layers) break else: llm_config.num_nextn_predict_layers = config.mtp_num_layers - HfConfigFactory.del_config_attr(hf_config, "quantization_config") + HfConfigFactory.del_config_attr(hf_config, 'quantization_config') expert_dtype = None - if ( - config.fp8 is not None - and config.fp8_recipe == "blockwise" - and config.fp8_param - ): - from transformers.utils.quantization_config import ( - FineGrainedFP8Config, - ) + if (config.fp8 is not None and config.fp8_recipe == 'blockwise' and config.fp8_param): + from transformers.utils.quantization_config import FineGrainedFP8Config modules_to_not_convert = get_modules_to_not_convert(self.hf_model) - if hasattr(self, "_fp8_skip_modules"): - modules_to_not_convert = (modules_to_not_convert or []) + list( - self._fp8_skip_modules - ) - hf_config.quantization_config = FineGrainedFP8Config( - modules_to_not_convert=modules_to_not_convert - ) - expert_dtype = "fp8" - if args.model_type == "deepseek_v4": - HfConfigFactory.set_config_attr( - hf_config, "expert_dtype", expert_dtype - ) + if hasattr(self, '_fp8_skip_modules'): + modules_to_not_convert = (modules_to_not_convert or []) + list(self._fp8_skip_modules) + hf_config.quantization_config = FineGrainedFP8Config(modules_to_not_convert=modules_to_not_convert) + expert_dtype = 'fp8' + if args.model_type == 'deepseek_v4': + HfConfigFactory.set_config_attr(hf_config, 'expert_dtype', expert_dtype) hf_config.save_pretrained(output_dir) - if getattr(self.hf_model, "_auto_class") is not None: + if getattr(self.hf_model, '_auto_class') is not None: try: custom_object_save(self.hf_model, output_dir, config=hf_config) except FileNotFoundError as e: @@ -420,16 +394,14 @@ def save_weights( model_dirs=[args.model_dir], additional_saved_files=self.hf_model.model_meta.additional_saved_files, ) - logger.info( - f"Successfully saved `safetensors` model weights in `{output_dir}`." - ) + logger.info(f"Successfully saved `safetensors` model weights in `{output_dir}`.") dist.barrier() # Ensure all weights are saved completely GPTBridge.save_weights = save_weights def init_megatron_env(): - os.environ.pop("VLLM_USE_MODELSCOPE", None) + os.environ.pop('VLLM_USE_MODELSCOPE', None) logging_level = logging.root.level _patch_unified_memory() _patch_transformers_output_recorder() @@ -439,11 +411,11 @@ def init_megatron_env(): try: _patch_torch_FileSystemReader() except Exception: - logger.warning("Failed to patch FileSystemReader.") + logger.warning('Failed to patch FileSystemReader.') try: _patch_validate_non_overlapping_shards_metadata() except Exception: - logger.warning("Patch validate_non_overlapping_shards_metadata failed.") + logger.warning('Patch validate_non_overlapping_shards_metadata failed.') pass import megatron.core diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index f7b2c1147d..ca387b0f12 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -37,15 +37,11 @@ def _check_padding_free(args, config): def _check_dsa_index_share_recompute(config): """Reject activation replay that omits a DSA skip layer's source indexer.""" - if ( - config.experimental_attention_variant == 'dsa' - and (getattr(config, 'dsa_indexer_topk_freq', 1) or 1) > 1 - and getattr(config, 'recompute_granularity', None) not in {None, 'none'} - ): + if (config.experimental_attention_variant == 'dsa' and (getattr(config, 'dsa_indexer_topk_freq', 1) or 1) > 1 + and getattr(config, 'recompute_granularity', None) not in {None, 'none'}): raise ValueError( 'DSA cross-layer top-k sharing is incompatible with activation recompute because a skip layer may be ' - 'replayed without its source computing layer. Set recompute_granularity=none.' - ) + 'replayed without its source computing layer. Set recompute_granularity=none.') def _get_hf_mtp_num_layers(hf_config): diff --git a/swift/megatron/trainers/trainer.py b/swift/megatron/trainers/trainer.py index 774bd55417..a7da085ef2 100644 --- a/swift/megatron/trainers/trainer.py +++ b/swift/megatron/trainers/trainer.py @@ -23,9 +23,8 @@ def project_owning_loader_semantics(input_values, model_label_values, semantic_l """Normalize padded Megatron carrier tensors back to the dataset semantic row.""" semantic_length = int(semantic_length) if semantic_length <= 0 or semantic_length > len(input_values) or semantic_length > len(model_label_values): - raise ValueError( - f'invalid owning-loader semantic length {semantic_length} for carrier lengths ' - f'{len(input_values)}/{len(model_label_values)}') + raise ValueError(f'invalid owning-loader semantic length {semantic_length} for carrier lengths ' + f'{len(input_values)}/{len(model_label_values)}') semantic_input_values = input_values[:semantic_length] normalized_label_values = model_label_values if labels_were_shifted and model_label_values: @@ -172,25 +171,32 @@ def _install_forward_contract_once(self): module_hits[module_name].append((chunk_index, module)) invalid = {name: len(hits) for name, hits in module_hits.items() if len(hits) != 1} if invalid: - raise RuntimeError(f'layer0 fine forward selectors must match exactly once on rank {rank}: {invalid}') + raise RuntimeError( + f'layer0 fine forward selectors must match exactly once on rank {rank}: {invalid}') for module_name, boundary in fine_specs.items(): chunk_index, module = module_hits[module_name][0] - handles.append(module.register_forward_hook( - lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) + handles.append( + module.register_forward_hook( + lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) selected.append({'chunk': chunk_index, 'module': module_name, 'boundary': boundary}) self._forward_contract_selector_receipt = selected rank_dir = os.path.join(output_dir, f'rank{rank}') os.makedirs(rank_dir, exist_ok=True) with open(os.path.join(rank_dir, 'metadata.json'), 'w', encoding='utf-8') as stream: - json.dump({ - 'schema': 'glm52-local-forward-boundaries/v1', - 'framework': 'torch', - 'rank': rank, - 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, - 'boundary_set': boundary_set, - 'selectors': selected, - 'records': {}, - }, stream, ensure_ascii=False, indent=2, sort_keys=True) + json.dump( + { + 'schema': 'glm52-local-forward-boundaries/v1', + 'framework': 'torch', + 'rank': rank, + 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, + 'boundary_set': boundary_set, + 'selectors': selected, + 'records': {}, + }, + stream, + ensure_ascii=False, + indent=2, + sort_keys=True) stream.write('\n') self._forward_contract_handles = handles self._forward_contract_installed = True @@ -205,22 +211,26 @@ def _install_forward_contract_once(self): local_layer = int(match.group(1)) global_layer = local_layer if rank < 2 else local_layer + 2 input_boundary = f'base_layer_{global_layer}_input' - handles.append(module.register_forward_pre_hook( - lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) + handles.append( + module.register_forward_pre_hook( + lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) boundary = f'base_layer_{global_layer}_output' elif module_name == 'decoder.final_layernorm': boundary = 'final_norm_output' elif module_name == 'output_layer': input_boundary = 'output_head_input' - handles.append(module.register_forward_pre_hook( - lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) + handles.append( + module.register_forward_pre_hook( + lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) boundary = 'output_head_output' elif module_name.startswith('mtp.layers.0.') and module_name.rsplit('.', 1)[-1] in { - 'enorm', 'hnorm', 'eh_proj', 'mtp_model_layer', 'layer_norm', 'final_layernorm'}: + 'enorm', 'hnorm', 'eh_proj', 'mtp_model_layer', 'layer_norm', 'final_layernorm' + }: boundary = f"mtp_{module_name.removeprefix('mtp.layers.0.').replace('.', '_')}_output" if boundary is not None: - handles.append(module.register_forward_hook( - lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) + handles.append( + module.register_forward_hook( + lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) self._forward_contract_handles = handles self._forward_contract_installed = True @@ -261,7 +271,8 @@ def _write_input_contract_once(self, data, seq_lens=None): return if not mpu.is_pipeline_last_stage(ignore_virtual=False): return - if torch.distributed.is_initialized() and torch.distributed.get_rank() != torch.distributed.get_world_size() - 1: + if (torch.distributed.is_initialized() + and torch.distributed.get_rank() != torch.distributed.get_world_size() - 1): return input_ids = data.get('input_ids') labels = data.get('labels') @@ -278,8 +289,8 @@ def digest(items): label_values = values(labels) model_mask_values = [label != -100 for label in label_values] semantic_length = seq_lens[0] if seq_lens else len(input_values) - labels_were_shifted = self.args.task_type == 'causal_lm' and not getattr( - self.args, 'pretokenized_dataset', False) + labels_were_shifted = self.args.task_type == 'causal_lm' and not getattr(self.args, 'pretokenized_dataset', + False) semantic_input_values, semantic_label_values, semantic_mask_values = project_owning_loader_semantics( input_values, label_values, semantic_length, labels_were_shifted) payload = { diff --git a/swift/megatron/utils/megatron_lm_utils.py b/swift/megatron/utils/megatron_lm_utils.py index b8d4618189..82ac657b2d 100644 --- a/swift/megatron/utils/megatron_lm_utils.py +++ b/swift/megatron/utils/megatron_lm_utils.py @@ -96,10 +96,8 @@ def configure_deterministic_mode(args): nccl_algo = os.environ.get('NCCL_ALGO') allowed_nccl_algorithms = {'Tree', 'Ring', 'CollnetDirect', 'CollnetChain', '^NVLS'} if nccl_algo not in allowed_nccl_algorithms: - raise ValueError( - f'NCCL_ALGO must be explicitly set to one of {sorted(allowed_nccl_algorithms)} ' - 'in deterministic mode.' - ) + raise ValueError(f'NCCL_ALGO must be explicitly set to one of {sorted(allowed_nccl_algorithms)} ' + 'in deterministic mode.') torch.use_deterministic_algorithms(True) torch.backends.cudnn.deterministic = True torch.backends.cudnn.benchmark = False diff --git a/tests/general/test_dataset_empty_assistant.py b/tests/general/test_dataset_empty_assistant.py index cff427ceba..2132de2bb0 100644 --- a/tests/general/test_dataset_empty_assistant.py +++ b/tests/general/test_dataset_empty_assistant.py @@ -4,12 +4,26 @@ def test_empty_assistant_filter_is_default_off(): - dataset = Dataset.from_list( - [ - {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": ""}]}, - {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": "a"}]}, - ] - ) + dataset = Dataset.from_list([ + { + 'messages': [{ + 'role': 'user', + 'content': 'q' + }, { + 'role': 'assistant', + 'content': '' + }] + }, + { + 'messages': [{ + 'role': 'user', + 'content': 'q' + }, { + 'role': 'assistant', + 'content': 'a' + }] + }, + ]) filtered = DatasetLoader()._filter_empty_assistant_responses(dataset) @@ -17,15 +31,37 @@ def test_empty_assistant_filter_is_default_off(): def test_empty_assistant_filter_drops_only_blank_assistant_content(): - dataset = Dataset.from_list( - [ - {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": ""}]}, - {"messages": [{"role": "user", "content": "q"}, {"role": "assistant", "content": " "}]}, - {"messages": [{"role": "user", "content": ""}, {"role": "assistant", "content": "a"}]}, - ] - ) + dataset = Dataset.from_list([ + { + 'messages': [{ + 'role': 'user', + 'content': 'q' + }, { + 'role': 'assistant', + 'content': '' + }] + }, + { + 'messages': [{ + 'role': 'user', + 'content': 'q' + }, { + 'role': 'assistant', + 'content': ' ' + }] + }, + { + 'messages': [{ + 'role': 'user', + 'content': '' + }, { + 'role': 'assistant', + 'content': 'a' + }] + }, + ]) filtered = DatasetLoader(drop_empty_assistant_response=True)._filter_empty_assistant_responses(dataset) assert len(filtered) == 1 - assert filtered[0]["messages"][-1]["content"] == "a" + assert filtered[0]['messages'][-1]['content'] == 'a' diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 2245461930..750745c986 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -127,9 +127,9 @@ def test_dsa_backend_forced_to_local_spec_when_accuracy_compatible(monkeypatch): def test_local_spec_mlp_norm_maps_pre_mlp_layernorm_when_unfused(): source = inspect.getsource(_patch_mcore_bridge_disable_te) - assert "fused_norm_weight is None" in source - assert "pre_mlp_layernorm.weight" in source - assert "mlp.linear_fc1.layer_norm_weight" in source + assert 'fused_norm_weight is None' in source + assert 'pre_mlp_layernorm.weight' in source + assert 'mlp.linear_fc1.layer_norm_weight' in source def test_dsa_index_share_rejects_selective_recompute(): diff --git a/tests/megatron/test_pretokenized_dataset.py b/tests/megatron/test_pretokenized_dataset.py index 523131d8d5..cc486241d2 100644 --- a/tests/megatron/test_pretokenized_dataset.py +++ b/tests/megatron/test_pretokenized_dataset.py @@ -1,8 +1,7 @@ -from types import SimpleNamespace - import pytest import torch from datasets import Dataset +from types import SimpleNamespace from swift.dataset import validate_pretokenized_dataset from swift.megatron.trainers import utils as trainer_utils @@ -51,6 +50,7 @@ def test_model_loader_uses_independent_processor_path(): requested = [] class FakeTokenizer: + @classmethod def from_pretrained(cls, path, **kwargs): requested.append((path, kwargs)) diff --git a/tests/megatron/test_raw_loss_observability.py b/tests/megatron/test_raw_loss_observability.py index 58c5f6a1f2..6de59899f1 100644 --- a/tests/megatron/test_raw_loss_observability.py +++ b/tests/megatron/test_raw_loss_observability.py @@ -6,19 +6,19 @@ def test_raw_loss_event_preserves_unrounded_values_and_step(): logs = { - "loss": 12.410510059999999, - "mtp_0_loss": 13.367947579999999, - "eval_loss": 99.0, + 'loss': 12.410510059999999, + 'mtp_0_loss': 13.367947579999999, + 'eval_loss': 99.0, } assert raw_loss_event(1, logs) == { - "step": 1, - "loss": 12.410510059999999, - "mtp_0_loss": 13.367947579999999, + 'step': 1, + 'loss': 12.410510059999999, + 'mtp_0_loss': 13.367947579999999, } def test_raw_loss_event_omits_non_loss_metrics(): - assert raw_loss_event(3, {"grad_norm": 1.0, "learning_rate": 1e-6}) is None + assert raw_loss_event(3, {'grad_norm': 1.0, 'learning_rate': 1e-6}) is None def test_owning_loader_projection_removes_sp_padding_and_reverses_label_roll(): @@ -36,37 +36,37 @@ def test_owning_loader_projection_rejects_invalid_semantic_length(): try: project_owning_loader_semantics([1, 2], [-100, 2], semantic_length=3) except ValueError as exc: - assert "invalid owning-loader semantic length" in str(exc) + assert 'invalid owning-loader semantic length' in str(exc) else: - raise AssertionError("invalid semantic length was accepted") + raise AssertionError('invalid semantic length was accepted') def test_parameter_record_preserves_orientation_and_signed_zero(): tensor = torch.tensor([[0.0, -0.0], [1.0, 2.0]], dtype=torch.float32) record = MegatronTrainer._parameter_record(tensor) - assert record["shape"] == [2, 2] - assert record["dtype"] == "torch.float32" - assert record["positive_zero_count"] == 1 - assert record["negative_zero_count"] == 1 - assert record["sha256"] != record["transpose_sha256"] + assert record['shape'] == [2, 2] + assert record['dtype'] == 'torch.float32' + assert record['positive_zero_count'] == 1 + assert record['negative_zero_count'] == 1 + assert record['sha256'] != record['transpose_sha256'] def test_layer0_fine_forward_specs_are_explicit_and_fail_closed(): - assert MegatronTrainer._forward_contract_specs("coarse") is None - specs = MegatronTrainer._forward_contract_specs("layer0_fine") + assert MegatronTrainer._forward_contract_specs('coarse') is None + specs = MegatronTrainer._forward_contract_specs('layer0_fine') assert len(specs) == 13 - assert specs["decoder.layers.0.input_layernorm"] == "layer0_input_rmsnorm_output" - assert specs["decoder.layers.0.self_attention.linear_q_down_proj"] == "layer0_q_down_projection_output" - assert specs["decoder.layers.0"] == "base_transformer_layer_0_output" + assert specs['decoder.layers.0.input_layernorm'] == 'layer0_input_rmsnorm_output' + assert specs['decoder.layers.0.self_attention.linear_q_down_proj'] == 'layer0_q_down_projection_output' + assert specs['decoder.layers.0'] == 'base_transformer_layer_0_output' try: - MegatronTrainer._forward_contract_specs("unknown") + MegatronTrainer._forward_contract_specs('unknown') except ValueError as exc: - assert "unsupported MODEL_REPRO_FORWARD_BOUNDARY_SET" in str(exc) + assert 'unsupported MODEL_REPRO_FORWARD_BOUNDARY_SET' in str(exc) else: - raise AssertionError("unknown forward boundary set was accepted") + raise AssertionError('unknown forward boundary set was accepted') def test_first_tensor_prefers_first_tensor_in_nested_module_output(): first = torch.tensor([1.0]) second = torch.tensor([2.0]) - assert MegatronTrainer._first_tensor((None, {"output": first}, second)) is first + assert MegatronTrainer._first_tensor((None, {'output': first}, second)) is first From 093872ca5d06e4e506abe3f2ecc41084d6f819c9 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 11:26:44 +0800 Subject: [PATCH 12/33] style: single-quote leftover strings in get_padding_to helpers CI pre-commit --all-files still ran double-quote-string-fixer on utils.py after the main merge. Local --all-files is now green. --- swift/megatron/utils/utils.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/swift/megatron/utils/utils.py b/swift/megatron/utils/utils.py index 4157033eb3..05e6226b71 100644 --- a/swift/megatron/utils/utils.py +++ b/swift/megatron/utils/utils.py @@ -84,11 +84,11 @@ def get_multimodal_target_regex( if not target_modules: continue target_modules = [tm for tm in target_modules if tm] - target_pattern = rf'.*\.({"|".join(target_modules)})' if target_modules else '' - rejected_pattern = rf'(?!({"|".join(rejected_modules)}))' if rejected_modules else '' + target_pattern = rf'.*\.({' | '.join(target_modules)})' if target_modules else '' + rejected_pattern = rf'(?!({' | '.join(rejected_modules)}))' if rejected_modules else '' res.append(rf'{rejected_pattern}{re.escape(module)}(?=\.){target_pattern}') - return rf'^({"|".join(res)})$' + return rf'^({' | '.join(res)})$' def get_target_modules(args, model): From 6a411aa426a52b9a92ae41b55854369d75b27ec2 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 11:28:07 +0800 Subject: [PATCH 13/33] style: keep regex '|' and quote dump helpers The previous quote-fixer pass rewrote rf'({"|".join(...)})' to ' | ', which would change the regex. Restore the original alternation and only single-quote the dump/load helper strings that CI's double-quote-string-fixer still rewrote. --- swift/megatron/utils/utils.py | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/swift/megatron/utils/utils.py b/swift/megatron/utils/utils.py index 05e6226b71..4577b06f78 100644 --- a/swift/megatron/utils/utils.py +++ b/swift/megatron/utils/utils.py @@ -84,11 +84,11 @@ def get_multimodal_target_regex( if not target_modules: continue target_modules = [tm for tm in target_modules if tm] - target_pattern = rf'.*\.({' | '.join(target_modules)})' if target_modules else '' - rejected_pattern = rf'(?!({' | '.join(rejected_modules)}))' if rejected_modules else '' + target_pattern = rf'.*\.({"|".join(target_modules)})' if target_modules else '' + rejected_pattern = rf'(?!({"|".join(rejected_modules)}))' if rejected_modules else '' res.append(rf'{rejected_pattern}{re.escape(module)}(?=\.){target_pattern}') - return rf'^({' | '.join(res)})$' + return rf'^({"|".join(res)})$' def get_target_modules(args, model): @@ -293,7 +293,7 @@ def get_load_fixed_data_path(): def _batch_data_suffix(step, rank, seq_len): - return f"step{step}_rank{rank}_seq{seq_len}.npy" + return f'step{step}_rank{rank}_seq{seq_len}.npy' def dump_batch_data(batch, step, seq_len): @@ -308,10 +308,10 @@ def dump_batch_data(batch, step, seq_len): torch.cuda.synchronize() os.makedirs(dump_path, exist_ok=True) suffix = _batch_data_suffix(step, rank, seq_len) - np.save(os.path.join(dump_path, f"tokens_{suffix}"), tokens.detach().cpu().numpy()) - np.save(os.path.join(dump_path, f"labels_{suffix}"), labels.detach().cpu().numpy()) + np.save(os.path.join(dump_path, f'tokens_{suffix}'), tokens.detach().cpu().numpy()) + np.save(os.path.join(dump_path, f'labels_{suffix}'), labels.detach().cpu().numpy()) if rank == 0: - print(f"[DUMP_DATA_PATH] saved tokens_{suffix} and labels_{suffix}", flush=True) + print(f'[DUMP_DATA_PATH] saved tokens_{suffix} and labels_{suffix}', flush=True) def load_fixed_batch_data(batch, step, seq_len): @@ -321,11 +321,11 @@ def load_fixed_batch_data(batch, step, seq_len): rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 suffix = _batch_data_suffix(step, rank, seq_len) - tokens_file = os.path.join(load_path, f"tokens_{suffix}") - labels_file = os.path.join(load_path, f"labels_{suffix}") + tokens_file = os.path.join(load_path, f'tokens_{suffix}') + labels_file = os.path.join(load_path, f'labels_{suffix}') if not (os.path.exists(tokens_file) and os.path.exists(labels_file)): if rank == 0: - print(f"[LOAD_FIXED_DATA_PATH] file not found: {tokens_file}", flush=True) + print(f'[LOAD_FIXED_DATA_PATH] file not found: {tokens_file}', flush=True) return batch tokens_np = np.load(tokens_file) From b0fc329c998303fe2d5324c0f0cb2b9d54e031eb Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 11:32:33 +0800 Subject: [PATCH 14/33] style: single-quote f-strings in megatron init for py3.10 CI CI lint uses Python 3.10, whose tokenize still sees f"..." as STRING tokens, so double-quote-string-fixer rewrites them. Local 3.12 tokenize splits f-strings and the hook no-ops. Convert the four logger f-strings and one comment on this branch. --- swift/megatron/init.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/swift/megatron/init.py b/swift/megatron/init.py index 943387db7a..721b811590 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -96,7 +96,7 @@ def validate_non_overlapping_shards_metadata(*args, **kwargs): api2.validate_non_overlapping_shards_metadata = (validate_non_overlapping_shards_metadata) def _validate_global_plan(*args, **kwargs): - # torch returns a list of error messages here; empty list means "no error". + # torch returns a list of error messages here; empty list means 'no error'. return [] default_planner._validate_global_plan = _validate_global_plan @@ -285,7 +285,7 @@ def _patch_mcore_bridge(): import mcore_bridge from mcore_bridge import GPTBridge from mcore_bridge.model.register import ModelLoader - logger.info(f"mcore_bridge.__version__: {mcore_bridge.__version__}") + logger.info(f'mcore_bridge.__version__: {mcore_bridge.__version__}') if _use_accuracy_compatible_enabled(): _patch_mcore_bridge_disable_te() if not getattr(ModelLoader._replace_spec_dsa, '_swift_norm_accuracy_patch', False): @@ -386,7 +386,7 @@ def save_weights( try: custom_object_save(self.hf_model, output_dir, config=hf_config) except FileNotFoundError as e: - logger.error(f"custom_object_save Error: {e}") + logger.error(f'custom_object_save Error: {e}') save_checkpoint( None, processor, @@ -394,7 +394,7 @@ def save_weights( model_dirs=[args.model_dir], additional_saved_files=self.hf_model.model_meta.additional_saved_files, ) - logger.info(f"Successfully saved `safetensors` model weights in `{output_dir}`.") + logger.info(f'Successfully saved `safetensors` model weights in `{output_dir}`.') dist.barrier() # Ensure all weights are saved completely GPTBridge.save_weights = save_weights @@ -419,4 +419,4 @@ def init_megatron_env(): pass import megatron.core - logger.info(f"megatron.core.__version__: {megatron.core.__version__}") + logger.info(f'megatron.core.__version__: {megatron.core.__version__}') From e47dafe7f8a7b5e4ebced3254647aeb3f97501a2 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 15:06:20 +0800 Subject: [PATCH 15/33] Skip passwordless-sudo ResetFileMode when no tty Self-hosted unittest died at sudo chown with "no tty present and no askpass". Keep chown when sudo -n works; otherwise skip so checkout can run. --- .github/workflows/citest.yaml | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/.github/workflows/citest.yaml b/.github/workflows/citest.yaml index 101e9f3b1a..12776e60e3 100644 --- a/.github/workflows/citest.yaml +++ b/.github/workflows/citest.yaml @@ -46,10 +46,16 @@ jobs: shell: bash run: | # reset filemode to allow action runner to delete files - # generated by root in docker + # generated by root in docker. Passwordless sudo is not guaranteed + # (self-hosted "sudo: no tty present and no askpass" failed #3). set -e - source ~/.bashrc - sudo chown -R $USER:$USER $GITHUB_WORKSPACE + source ~/.bashrc || true + if command -v sudo >/dev/null 2>&1 && sudo -n true 2>/dev/null; then + sudo -n chown -R "$USER:$USER" "$GITHUB_WORKSPACE" + else + echo "skip sudo chown: no passwordless sudo" + chown -R "$USER:$USER" "$GITHUB_WORKSPACE" 2>/dev/null || true + fi - name: Checkout uses: actions/checkout@v3 From 73a700dd78d4962f0727e63df812edf49bffaa8a Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 16:12:56 +0800 Subject: [PATCH 16/33] Select stack-paired PaddleFleet pin for alignment CI Keep pull_request on the historical CodeSync/develop tarball and develop/latest wheels. workflow_dispatch can opt into fail-closed SHA checkout plus artifact digest checks; pairing remains unproven unless this job built the wheels from the pin. MinimaxV2.5_EP2 and GLM45Air_EP2 are unchanged. --- .../workflows/alignment_model_accuracy.yaml | 103 ++- scripts/select_paddlefleet_alignment_pin.sh | 638 ++++++++++++++++++ 2 files changed, 724 insertions(+), 17 deletions(-) create mode 100755 scripts/select_paddlefleet_alignment_pin.sh diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index 6de447d2e4..b06448ec12 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -3,6 +3,36 @@ name: Alignment Model Accuracy on: pull_request: workflow_dispatch: + inputs: + paddlefleet_mode: + description: "develop keeps the historical CodeSync tarball. stack-paired fail-closed SHA check." + type: choice + default: develop + options: [develop, stack-paired] + paddlefleet_pin_sha: + description: "40-hex PaddleFleet commit (required for stack-paired)" + required: false + type: string + paddlefleet_git_url: + description: "Git URL that contains paddlefleet_pin_sha" + required: false + type: string + paddlefleet_wheel_url: + description: "Optional wheel URL from Build Fleet whl / Actions artifact metadata" + required: false + type: string + paddlefleet_wheel_sha256: + description: "sha256 for paddlefleet_wheel_url" + required: false + type: string + paddlefleet_ops_wheel_url: + description: "Optional paddlefleet_ops wheel URL" + required: false + type: string + paddlefleet_ops_wheel_sha256: + description: "sha256 for paddlefleet_ops_wheel_url" + required: false + type: string concurrency: group: Alignment-${{ github.workflow }}-${{ github.event.pull_request.number }} @@ -17,6 +47,14 @@ env: TASK: MS-SWIFT-${{ github.sha }}-alignment CE_name: alignment-ms-swift no_proxy: "localhost,bj.bcebos.com,su.bcebos.com,bcebos.com,apiin.im.baidu.com,gitee.com,aliyun.com,.baidu.com,.tuna.tsinghua.edu.cn" + # pull_request stays on historical develop. stack-paired is workflow_dispatch only. + ALIGNMENT_PADDLEFLEET_MODE: ${{ github.event_name == 'workflow_dispatch' && github.event.inputs.paddlefleet_mode || 'develop' }} + PADDLEFLEET_PIN_SHA: ${{ github.event.inputs.paddlefleet_pin_sha }} + PADDLEFLEET_GIT_URL: ${{ github.event.inputs.paddlefleet_git_url }} + PADDLEFLEET_WHEEL_URL: ${{ github.event.inputs.paddlefleet_wheel_url }} + PADDLEFLEET_WHEEL_SHA256: ${{ github.event.inputs.paddlefleet_wheel_sha256 }} + PADDLEFLEET_OPS_WHEEL_URL: ${{ github.event.inputs.paddlefleet_ops_wheel_url }} + PADDLEFLEET_OPS_WHEEL_SHA256: ${{ github.event.inputs.paddlefleet_ops_wheel_sha256 }} defaults: run: @@ -55,16 +93,27 @@ jobs: -e no_proxy \ -e CE_name \ -e python_version \ + -e ALIGNMENT_PADDLEFLEET_MODE \ + -e PADDLEFLEET_PIN_SHA \ + -e PADDLEFLEET_GIT_URL \ + -e PADDLEFLEET_WHEEL_URL \ + -e PADDLEFLEET_WHEEL_SHA256 \ + -e PADDLEFLEET_OPS_WHEEL_URL \ + -e PADDLEFLEET_OPS_WHEEL_SHA256 \ -w /workspace $IMAGE_NAME - name: Checkout Code run: | docker exec -t $container_name /bin/bash -c ' rm -rf * .[^.]* source $work_dir/../../../proxy - echo "Download PaddleFleet form https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" - wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate - rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar - cd PaddleFleet && git pull && cd - + if [ "${ALIGNMENT_PADDLEFLEET_MODE:-develop}" = "stack-paired" ]; then + echo "stack-paired: skip CodeSync/develop PaddleFleet.tar; selector runs after python" + else + echo "Download PaddleFleet form https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" + wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate + rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar + cd PaddleFleet && git pull && cd - + fi echo "Download ms-swift form https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz" wget -q --no-proxy https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate @@ -110,18 +159,34 @@ jobs: ldconfig BOS=https://paddle-github-action.bj.bcebos.com - echo "::group::Download paddlefleet / paddlefleet_ops / megatron-core / mcore-bridge wheels from BOS" cd /workspace - for url in \ - $BOS/PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl \ - $BOS/PaddleFleet/develop/latest/cu130/paddle-release/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl \ - $BOS/whl/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl \ - $BOS/whl/mcore_bridge-0.0.0-py3-none-any.whl ; do - echo "Downloading $url" - wget -q --no-proxy --no-check-certificate --tries=3 --timeout=60 "$url" - done + if [ "${ALIGNMENT_PADDLEFLEET_MODE:-develop}" = "stack-paired" ]; then + echo "::group::Select stack-paired PaddleFleet (fail-closed SHA)" + python -m pip install uv + bash /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh --dest /workspace + cat /workspace/paddlefleet_alignment_pin_receipt.json + echo "::endgroup::" + echo "::group::Download remaining wheels from BOS" + for url in \ + $BOS/whl/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl \ + $BOS/whl/mcore_bridge-0.0.0-py3-none-any.whl ; do + echo "Downloading $url" + wget -q --no-proxy --no-check-certificate --tries=3 --timeout=60 "$url" + done + echo "::endgroup::" + else + echo "::group::Download paddlefleet / paddlefleet_ops / megatron-core / mcore-bridge wheels from BOS" + for url in \ + $BOS/PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl \ + $BOS/PaddleFleet/develop/latest/cu130/paddle-release/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl \ + $BOS/whl/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl \ + $BOS/whl/mcore_bridge-0.0.0-py3-none-any.whl ; do + echo "Downloading $url" + wget -q --no-proxy --no-check-certificate --tries=3 --timeout=60 "$url" + done + echo "::endgroup::" + fi ls -l /workspace/ - echo "::endgroup::" echo "::group::Build ms-swift wheel" cd /workspace/ms-swift @@ -140,8 +205,12 @@ jobs: source $work_dir/../../../proxy export PROXY_URL="${http_proxy}" - export PADDLEFLEET_WHEEL_PATH="/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" - export PADDLEFLEET_OPS_WHEEL_PATH="/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" + if [ -f /workspace/paddlefleet_alignment_pin.env ]; then + . /workspace/paddlefleet_alignment_pin.env + echo "loaded pin receipt ${PADDLEFLEET_PIN_RECEIPT:-/workspace/paddlefleet_alignment_pin_receipt.json}" + fi + export PADDLEFLEET_WHEEL_PATH="${PADDLEFLEET_WHEEL_PATH:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" + export PADDLEFLEET_OPS_WHEEL_PATH="${PADDLEFLEET_OPS_WHEEL_PATH:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" export MEGATRON_CORE_WHEEL_PATH=/workspace/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl export MS_SWIFT_WHEEL_PATH=/workspace/upload/ms_swift-0.0.0-py3-none-any.whl export MCORE_BRIDGE_WHEEL_PATH=/workspace/mcore_bridge-0.0.0-py3-none-any.whl @@ -151,7 +220,7 @@ jobs: for whl in "$PADDLEFLEET_WHEEL_PATH" "$PADDLEFLEET_OPS_WHEEL_PATH" \ "$MS_SWIFT_WHEEL_PATH" "$MEGATRON_CORE_WHEEL_PATH" \ "$MCORE_BRIDGE_WHEEL_PATH"; do - [ -f "$whl" ] || { echo "::error:: missing wheel: $whl"; exit 1; } + [ -f "$whl" ] || [ -d "$whl" ] || { echo "::error:: missing wheel: $whl"; exit 1; } echo "using $whl" done python -m pip install uv diff --git a/scripts/select_paddlefleet_alignment_pin.sh b/scripts/select_paddlefleet_alignment_pin.sh new file mode 100755 index 0000000000..71c2131318 --- /dev/null +++ b/scripts/select_paddlefleet_alignment_pin.sh @@ -0,0 +1,638 @@ +#!/usr/bin/env bash +# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Select PaddleFleet source + wheels for alignment_model_accuracy. +# +# Default (ALIGNMENT_PADDLEFLEET_MODE=develop or unset): +# historical CodeSync/develop tarball + BOS develop/latest wheels. +# Cases are not filtered. +# +# Explicit (ALIGNMENT_PADDLEFLEET_MODE=stack-paired): fail-closed. +# Checkout PADDLEFLEET_PIN_SHA; git rev-parse HEAD must equal the pin. +# Artifacts: caller URL+sha256 (from Build Fleet whl / Actions metadata) +# or build from the checked-out tree. Independent digest matches do not +# prove the wheels were produced from that commit — receipt records +# source_commit vs artifact sha256 separately and pairing as unproven +# unless this invocation built the files from the pin. +# git/download/checkout failures still write an error receipt. + +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: select_paddlefleet_alignment_pin.sh [--dest DIR] [--self-test] + +--self-test ignores --dest and uses an offline fixture (no network). + +Env: + ALIGNMENT_PADDLEFLEET_MODE develop (default) | stack-paired + PADDLEFLEET_PIN_SHA required 40-hex commit in stack-paired + PADDLEFLEET_GIT_URL git remote or local repo (stack-paired) + PADDLEFLEET_WHEEL_URL optional explicit wheel (https or local path) + PADDLEFLEET_WHEEL_SHA256 required with WHEEL_URL + PADDLEFLEET_OPS_WHEEL_URL optional explicit ops wheel + PADDLEFLEET_OPS_WHEEL_SHA256 required with OPS URL + PADDLEFLEET_BUILD_CMD optional; default uv build paddlefleet + PADDLEFLEET_BUILD_OPS_CMD optional; default uv build paddlefleet-ops + ALIGNMENT_PADDLEFLEET_DEST output directory (default /workspace) +EOF +} + +MODE="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" +DEST="${ALIGNMENT_PADDLEFLEET_DEST:-/workspace}" +RUN_SELF_TEST=0 +while [[ $# -gt 0 ]]; do + case "$1" in + --dest) + DEST="${2:?--dest requires a path}" + shift 2 + ;; + --self-test) + RUN_SELF_TEST=1 + shift + ;; + -h|--help) + usage + exit 0 + ;; + *) + echo "unknown arg: $1" >&2 + usage + exit 2 + ;; + esac +done + +BOS="${PADDLEFLEET_BOS:-https://paddle-github-action.bj.bcebos.com}" +DEFAULT_TAR_URL="https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" +DEFAULT_WHL_URL="${BOS}/PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" +DEFAULT_OPS_URL="${BOS}/PaddleFleet/develop/latest/cu130/paddle-release/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" +GIT_URL="${PADDLEFLEET_GIT_URL:-https://github.com/PaddlePaddle/PaddleFleet.git}" +PIN_SHA="${PADDLEFLEET_PIN_SHA:-}" +WHEEL_URL="${PADDLEFLEET_WHEEL_URL:-}" +WHEEL_SHA="${PADDLEFLEET_WHEEL_SHA256:-}" +OPS_URL="${PADDLEFLEET_OPS_WHEEL_URL:-}" +OPS_SHA="${PADDLEFLEET_OPS_WHEEL_SHA256:-}" +BUILD_CMD="${PADDLEFLEET_BUILD_CMD:-}" +BUILD_OPS_CMD="${PADDLEFLEET_BUILD_OPS_CMD:-}" + +ACTUAL_SHA="" +SOURCE_VERIFIED=false +PADDLEFLEET_WHEEL_PATH="" +PADDLEFLEET_OPS_WHEEL_PATH="" +ACTUAL_WHEEL_SHA="" +ACTUAL_OPS_SHA="" +WHEEL_DIGEST_VERIFIED=false +OPS_DIGEST_VERIFIED=false +WHEEL_ORIGIN="" +OPS_ORIGIN="" +WHEEL_BUILT_FROM_COMMIT="" +OPS_BUILT_FROM_COMMIT="" +LOADED_FROM="" +RECEIPT_WRITTEN=0 + +log() { echo "[paddlefleet-pin] $*" >&2; } + +sha256_file() { sha256sum -- "$1" | awk '{print $1}'; } + +unpaired_url() { + case "$1" in + *"/develop/latest/"*|*"CodeSync/develop/"*) return 0 ;; + *) return 1 ;; + esac +} + +pairing_fields() { + local proven=false + local status="unproven" + local reason="wheel/ops digest match does not prove production from source_commit" + if [[ "${MODE}" == "develop" ]]; then + status="unpaired_default" + reason="develop tarball and develop/latest wheels; not a stack pin" + elif [[ "${WHEEL_ORIGIN}" == "build" && "${OPS_ORIGIN}" == "build" \ + && "${WHEEL_BUILT_FROM_COMMIT}" == "${ACTUAL_SHA}" \ + && "${OPS_BUILT_FROM_COMMIT}" == "${ACTUAL_SHA}" \ + && "${SOURCE_VERIFIED}" == "true" ]]; then + status="built_from_checked_out_pin" + reason="this invocation built both artifacts from checked-out source_commit; not a remote-stack proof" + fi + printf '%s\t%s\t%s\n' "${proven}" "${status}" "${reason}" +} + +write_receipt() { + local status="$1" detail="${2:-}" + mkdir -p "${DEST}" + local receipt="${DEST}/paddlefleet_alignment_pin_receipt.json" + local pair + pair="$(pairing_fields)" + local stack_proven pairing_status pairing_reason + stack_proven="${pair%%$'\t'*}" + pair="${pair#*$'\t'}" + pairing_status="${pair%%$'\t'*}" + pairing_reason="${pair#*$'\t'}" + if ! command -v python3 >/dev/null 2>&1; then + printf '{"schema":"paddlefleet-alignment-pin/v1","status":"%s","detail":"%s"}\n' \ + "${status}" "${detail}" >"${receipt}" + RECEIPT_WRITTEN=1 + return 0 + fi + python3 - "${receipt}" "${status}" "${detail}" "${stack_proven}" \ + "${pairing_status}" "${pairing_reason}" <<'PY' +import json, os, sys +from datetime import datetime, timezone +path, status, detail, stack_proven, pairing_status, pairing_reason = sys.argv[1:7] + +def art(name, pth, url, exp, act, digest_ok, origin, built_from): + if not pth: + return None + return { + "name": name, + "path": pth, + "url": url or None, + "expected_sha256": exp or None, + "actual_sha256": act or None, + "digest_verified": digest_ok == "true", + "origin": origin or None, + "built_from_commit": built_from or None, + } + +arts = [a for a in ( + art("paddlefleet", os.environ.get("PADDLEFLEET_WHEEL_PATH", ""), + os.environ.get("WHEEL_URL", ""), os.environ.get("WHEEL_SHA", ""), + os.environ.get("ACTUAL_WHEEL_SHA", ""), os.environ.get("WHEEL_DIGEST_VERIFIED", "false"), + os.environ.get("WHEEL_ORIGIN", ""), os.environ.get("WHEEL_BUILT_FROM_COMMIT", "")), + art("paddlefleet_ops", os.environ.get("PADDLEFLEET_OPS_WHEEL_PATH", ""), + os.environ.get("OPS_URL", ""), os.environ.get("OPS_SHA", ""), + os.environ.get("ACTUAL_OPS_SHA", ""), os.environ.get("OPS_DIGEST_VERIFIED", "false"), + os.environ.get("OPS_ORIGIN", ""), os.environ.get("OPS_BUILT_FROM_COMMIT", "")), +) if a] + +doc = { + "schema": "paddlefleet-alignment-pin/v1", + "status": status, + "detail": detail, + "captured_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "mode": os.environ.get("MODE"), + "dest": os.environ.get("DEST"), + "source": { + "git_url": os.environ.get("GIT_URL") or None, + "expected_commit": os.environ.get("PIN_SHA") or None, + "actual_commit": os.environ.get("ACTUAL_SHA") or None, + "commit_verified": os.environ.get("SOURCE_VERIFIED") == "true", + }, + "loaded_from": os.environ.get("LOADED_FROM") or None, + "artifacts": arts, + "pairing": { + "stack_paired_proven": stack_proven == "true", + "status": pairing_status, + "reason": pairing_reason, + }, + "default_urls": { + "source_tar": os.environ.get("DEFAULT_TAR_URL"), + "paddlefleet_wheel": os.environ.get("DEFAULT_WHL_URL"), + "paddlefleet_ops_wheel": os.environ.get("DEFAULT_OPS_URL"), + }, + "cases_preserved": ["MinimaxV2.5_EP2", "GLM45Air_EP2"], + "unpaired_develop_rejected_in_stack_paired": True, +} +open(path, "w", encoding="utf-8").write(json.dumps(doc, indent=2) + "\n") +print("[paddlefleet-pin] receipt", path, file=sys.stderr) +PY + RECEIPT_WRITTEN=1 +} + +export_receipt_env() { + export MODE DEST GIT_URL PIN_SHA ACTUAL_SHA SOURCE_VERIFIED LOADED_FROM + export PADDLEFLEET_WHEEL_PATH PADDLEFLEET_OPS_WHEEL_PATH + export WHEEL_URL WHEEL_SHA ACTUAL_WHEEL_SHA WHEEL_DIGEST_VERIFIED WHEEL_ORIGIN WHEEL_BUILT_FROM_COMMIT + export OPS_URL OPS_SHA ACTUAL_OPS_SHA OPS_DIGEST_VERIFIED OPS_ORIGIN OPS_BUILT_FROM_COMMIT + export DEFAULT_TAR_URL DEFAULT_WHL_URL DEFAULT_OPS_URL +} + +fail() { + trap - ERR + local msg="$1" + log "FAIL: ${msg}" + export_receipt_env + write_receipt "error" "${msg}" + echo "::error:: ${msg}" >&2 + exit 1 +} + +on_err() { + local rc=$? + if [[ "${RECEIPT_WRITTEN}" == 1 || "${RUN_SELF_TEST}" == 1 ]]; then + return "${rc}" + fi + fail "command failed rc=${rc}" +} +trap 'on_err' ERR + +write_envfile() { + cat >"${DEST}/paddlefleet_alignment_pin.env" < ${out}" + mkdir -p "$(dirname "${out}")" + if [[ "${url}" == file://* ]]; then + local src="${url#file://}" + [[ -f "${src}" ]] || fail "download failed, local file missing: ${src}" + cp -- "${src}" "${out}" || fail "download copy failed: ${src}" + return 0 + fi + if [[ "${url}" == /* ]]; then + [[ -f "${url}" ]] || fail "download failed, local file missing: ${url}" + cp -- "${url}" "${out}" || fail "download copy failed: ${url}" + return 0 + fi + if wget -q --no-proxy --no-check-certificate --tries=2 --timeout=15 -O "${out}" "${url}"; then + return 0 + fi + fail "download failed: ${url}" +} + +# Must not run inside $(); fail() has to exit this shell. +require_digest() { + local path="$1" expected="$2" label="$3" actual="$4" + [[ -f "${path}" ]] || fail "missing ${label}: ${path}" + [[ -n "${expected}" ]] || fail "stack-paired missing ${label} sha256" + if [[ "${actual}" != "${expected}" ]]; then + fail "stack-paired ${label} sha256 mismatch expected=${expected} actual=${actual}" + fi +} + +checkout_pin() { + [[ "${PIN_SHA}" =~ ^[0-9a-fA-F]{40}$ ]] || fail "stack-paired requires PADDLEFLEET_PIN_SHA (40 hex), got '${PIN_SHA}'" + PIN_SHA="$(printf '%s' "${PIN_SHA}" | tr 'A-F' 'a-f')" + rm -rf "${DEST}/PaddleFleet" + log "clone ${GIT_URL}" + if ! git clone --quiet "${GIT_URL}" "${DEST}/PaddleFleet" >/dev/null 2>"${DEST}/.git-clone.err"; then + fail "git clone failed: $(tr '\n' ' ' <"${DEST}/.git-clone.err")" + fi + git -C "${DEST}/PaddleFleet" config advice.detachedHead false || true + log "checkout ${PIN_SHA}" + if ! git -C "${DEST}/PaddleFleet" checkout --quiet --force "${PIN_SHA}" >/dev/null 2>"${DEST}/.git-co.err"; then + fail "git checkout failed for ${PIN_SHA}: $(tr '\n' ' ' <"${DEST}/.git-co.err")" + fi + ACTUAL_SHA="$(git -C "${DEST}/PaddleFleet" rev-parse HEAD)" + if [[ "${ACTUAL_SHA}" != "${PIN_SHA}" ]]; then + SOURCE_VERIFIED=false + fail "stack-paired source SHA mismatch expected=${PIN_SHA} actual=${ACTUAL_SHA}" + fi + SOURCE_VERIFIED=true + log "source commit verified ${ACTUAL_SHA}" +} + +# Sets DEST_PATH and DEST_SHA in the caller. Must run in this shell so +# fail() writes the receipt (never wrap this in $()). +acquire_explicit() { + local url="$1" expected="$2" dest_name="$3" label="$4" + unpaired_url "${url}" && fail "stack-paired rejects unpaired ${label} URL: ${url}" + [[ -n "${expected}" ]] || fail "stack-paired ${label} URL requires matching sha256" + download "${url}" "${DEST}/${dest_name}" + DEST_PATH="${DEST}/${dest_name}" + DEST_SHA="$(sha256_file "${DEST_PATH}")" + # Record path/digest before require_digest so a mismatch receipt still has them. + if [[ "${label}" == paddlefleet\ wheel ]]; then + PADDLEFLEET_WHEEL_PATH="${DEST_PATH}" + ACTUAL_WHEEL_SHA="${DEST_SHA}" + WHEEL_ORIGIN="ci_metadata" + else + PADDLEFLEET_OPS_WHEEL_PATH="${DEST_PATH}" + ACTUAL_OPS_SHA="${DEST_SHA}" + OPS_ORIGIN="ci_metadata" + fi + require_digest "${DEST_PATH}" "${expected}" "${label}" "${DEST_SHA}" + LOADED_FROM="${LOADED_FROM:+${LOADED_FROM};}${DEST_PATH} from ${url}" +} + +run_build() { + local cmd="$1" glob="$2" label="$3" + mkdir -p "${DEST}/dist" + log "build ${label}: ${cmd}" + if ! (cd "${DEST}/PaddleFleet" && bash -lc "${cmd}"); then + fail "stack-paired build failed for ${label}" + fi + local built + built="$(ls -1 ${glob} 2>/dev/null | head -n 1 || true)" + [[ -n "${built}" && -f "${built}" ]] || fail "stack-paired build produced no ${label} (glob ${glob})" + local dest_name + dest_name="$(basename "${built}")" + cp -f -- "${built}" "${DEST}/${dest_name}" + DEST_PATH="${DEST}/${dest_name}" + DEST_SHA="$(sha256_file "${DEST_PATH}")" + LOADED_FROM="${LOADED_FROM:+${LOADED_FROM};}built ${DEST_PATH} from ${ACTUAL_SHA}" +} + +acquire_wheel() { + if [[ -n "${WHEEL_URL}" ]]; then + acquire_explicit "${WHEEL_URL}" "${WHEEL_SHA}" "paddlefleet.whl" "paddlefleet wheel" + PADDLEFLEET_WHEEL_PATH="${DEST_PATH}" + ACTUAL_WHEEL_SHA="${DEST_SHA}" + WHEEL_DIGEST_VERIFIED=true + WHEEL_ORIGIN="ci_metadata" + WHEEL_BUILT_FROM_COMMIT="" + else + local cmd="${BUILD_CMD:-uv build --wheel --package paddlefleet --out-dir '${DEST}/dist' --clear}" + run_build "${cmd}" "${DEST}/dist/paddlefleet-*.whl" "paddlefleet wheel" + PADDLEFLEET_WHEEL_PATH="${DEST_PATH}" + ACTUAL_WHEEL_SHA="${DEST_SHA}" + WHEEL_DIGEST_VERIFIED=true + WHEEL_ORIGIN="build" + WHEEL_BUILT_FROM_COMMIT="${ACTUAL_SHA}" + fi +} + +acquire_ops() { + if [[ -n "${OPS_URL}" ]]; then + acquire_explicit "${OPS_URL}" "${OPS_SHA}" "paddlefleet_ops.whl" "paddlefleet_ops wheel" + PADDLEFLEET_OPS_WHEEL_PATH="${DEST_PATH}" + ACTUAL_OPS_SHA="${DEST_SHA}" + OPS_DIGEST_VERIFIED=true + OPS_ORIGIN="ci_metadata" + OPS_BUILT_FROM_COMMIT="" + else + local cmd="${BUILD_OPS_CMD:-uv build --wheel --package paddlefleet-ops --out-dir '${DEST}/dist' --no-build-isolation}" + run_build "${cmd}" "${DEST}/dist/paddlefleet_ops-*.whl" "paddlefleet_ops wheel" + PADDLEFLEET_OPS_WHEEL_PATH="${DEST_PATH}" + ACTUAL_OPS_SHA="${DEST_SHA}" + OPS_DIGEST_VERIFIED=true + OPS_ORIGIN="build" + OPS_BUILT_FROM_COMMIT="${ACTUAL_SHA}" + fi +} + +fetch_default() { + log "mode=develop (historical unpaired CodeSync tarball)" + download "${DEFAULT_TAR_URL}" "${DEST}/PaddleFleet.tar" + rm -rf "${DEST}/PaddleFleet" + tar xf "${DEST}/PaddleFleet.tar" -C "${DEST}" + rm -f "${DEST}/PaddleFleet.tar" + if [[ -d "${DEST}/PaddleFleet/.git" ]]; then + git -C "${DEST}/PaddleFleet" pull || log "git pull skipped" + ACTUAL_SHA="$(git -C "${DEST}/PaddleFleet" rev-parse HEAD 2>/dev/null || true)" + fi + download "${DEFAULT_WHL_URL}" "${DEST}/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" + download "${DEFAULT_OPS_URL}" "${DEST}/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" + PADDLEFLEET_WHEEL_PATH="${DEST}/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" + PADDLEFLEET_OPS_WHEEL_PATH="${DEST}/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" + WHEEL_URL="${DEFAULT_WHL_URL}" + OPS_URL="${DEFAULT_OPS_URL}" + ACTUAL_WHEEL_SHA="$(sha256_file "${PADDLEFLEET_WHEEL_PATH}")" + ACTUAL_OPS_SHA="$(sha256_file "${PADDLEFLEET_OPS_WHEEL_PATH}")" + WHEEL_ORIGIN="develop_latest" + OPS_ORIGIN="develop_latest" + SOURCE_VERIFIED=false + WHEEL_DIGEST_VERIFIED=false + OPS_DIGEST_VERIFIED=false + LOADED_FROM="develop_latest ${DEFAULT_WHL_URL} ${DEFAULT_OPS_URL}" + export_receipt_env + write_envfile + write_receipt "ok" "develop tarball and develop/latest wheels; unpaired with a stack pin" +} + +fetch_stack_paired() { + log "mode=stack-paired" + checkout_pin + acquire_wheel + acquire_ops + export_receipt_env + write_envfile + local pair pairing_status + pair="$(pairing_fields)" + pair="${pair#*$'\t'}" + pairing_status="${pair%%$'\t'*}" + write_receipt "ok" "source_commit checked out; pairing.status=${pairing_status}; stack_paired_proven=false" +} + +install_offline_stubs() { + local bin="$1" + mkdir -p "${bin}" + cat >"${bin}/wget" <<'WGET' +#!/usr/bin/env bash +out="" +url="" +while [[ $# -gt 0 ]]; do + case "$1" in + -O) out="$2"; shift 2 ;; + --*) shift ;; + *) url="$1"; shift ;; + esac +done +if [[ -z "${url}" || -z "${out}" ]]; then + echo "wget-stub: missing url/out" >&2 + exit 1 +fi +if [[ "${url}" == http://* || "${url}" == https://* ]]; then + echo "wget-stub: blocked network ${url}" >&2 + exit 1 +fi +src="${url#file://}" +if [[ -f "${src}" ]]; then + cp -- "${src}" "${out}" + exit 0 +fi +echo "wget-stub: not a local file ${url}" >&2 +exit 1 +WGET + chmod +x "${bin}/wget" +} + +run_self_test() { + trap - ERR + local root script + root="$(mktemp -d)" + script="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)/$(basename -- "${BASH_SOURCE[0]}")" + trap 'rm -rf "${root}"' RETURN + install_offline_stubs "${root}/bin" + export PATH="${root}/bin:${PATH}" + + git init -q "${root}/upstream" + git -C "${root}/upstream" config user.email test@example.com + git -C "${root}/upstream" config user.name test + echo source-a >"${root}/upstream/README" + git -C "${root}/upstream" add README + git -C "${root}/upstream" commit -q -m a + local sha_a sha_b + sha_a="$(git -C "${root}/upstream" rev-parse HEAD)" + echo source-b >"${root}/upstream/README" + git -C "${root}/upstream" add README + git -C "${root}/upstream" commit -q -m b + sha_b="$(git -C "${root}/upstream" rev-parse HEAD)" + + mkdir -p "${root}/art" + echo py-body >"${root}/art/py.whl" + echo ops-body >"${root}/art/ops.whl" + local py_sha ops_sha + py_sha="$(sha256_file "${root}/art/py.whl")" + ops_sha="$(sha256_file "${root}/art/ops.whl")" + + expect_fail() { + local dest="$1" + local needle="$2" + shift 2 + mkdir -p "${dest}" + if "$@"; then + echo "self-test FAIL: expected failure (${needle})" >&2 + exit 1 + fi + local rec="${dest}/paddlefleet_alignment_pin_receipt.json" + [[ -f "${rec}" ]] || { echo "self-test FAIL: missing error receipt ${rec}" >&2; exit 1; } + grep -q '"status": "error"' "${rec}" + grep -q "${needle}" "${rec}" + echo "[self-test] fail-closed ${dest}: ${needle}" + } + + local run + run() { env PATH="${root}/bin:${PATH}" "$@"; } + + expect_fail "${root}/m1" "PADDLEFLEET_PIN_SHA" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA= \ + bash "${script}" --dest "${root}/m1" + + expect_fail "${root}/m2" "rejects unpaired" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + PADDLEFLEET_WHEEL_URL="${DEFAULT_WHL_URL}" \ + PADDLEFLEET_WHEEL_SHA256="${py_sha}" \ + PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ + PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ + bash "${script}" --dest "${root}/m2" + + expect_fail "${root}/m3" "git checkout failed" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="0000000000000000000000000000000000000000" \ + PADDLEFLEET_GIT_URL="${root}/upstream" \ + bash "${script}" --dest "${root}/m3" + + # Real checksum mismatch after a successful local copy (not a wget miss). + expect_fail "${root}/m4" "sha256 mismatch" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + PADDLEFLEET_WHEEL_URL="${root}/art/py.whl" \ + PADDLEFLEET_WHEEL_SHA256="deadbeefdeadbeefdeadbeefdeadbeefdeadbeefdeadbeefdeadbeefdeadbeef" \ + PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ + PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ + bash "${script}" --dest "${root}/m4" + python3 - "${root}/m4/paddlefleet_alignment_pin_receipt.json" "${py_sha}" <<'PY' +import json, sys +doc = json.load(open(sys.argv[1])) +assert doc["status"] == "error" +wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") +assert wheel["actual_sha256"] == sys.argv[2] +assert wheel["digest_verified"] is False +assert wheel["actual_sha256"] != (wheel.get("expected_sha256") or "") +print("m4 checksum-mismatch receipt has actual digest, not a download miss") +PY + + expect_fail "${root}/m5" "produced no paddlefleet" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + PADDLEFLEET_BUILD_CMD="mkdir -p '${root}/m5/dist'" \ + PADDLEFLEET_BUILD_OPS_CMD="true" \ + bash "${script}" --dest "${root}/m5" + + expect_fail "${root}/m6" "download failed" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + PADDLEFLEET_WHEEL_URL="https://example.invalid/paddlefleet.whl" \ + PADDLEFLEET_WHEEL_SHA256="${py_sha}" \ + PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ + PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ + bash "${script}" --dest "${root}/m6" + + expect_fail "${root}/m7" "git clone failed" \ + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" \ + PADDLEFLEET_GIT_URL="${root}/no-such-remote" \ + bash "${script}" --dest "${root}/m7" + + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + PADDLEFLEET_WHEEL_URL="${root}/art/py.whl" \ + PADDLEFLEET_WHEEL_SHA256="${py_sha}" \ + PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ + PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ + bash "${script}" --dest "${root}/ok-url" + python3 - "${root}/ok-url/paddlefleet_alignment_pin_receipt.json" "${sha_b}" "${py_sha}" <<'PY' +import json, sys +doc = json.load(open(sys.argv[1])) +sha_b, py_sha = sys.argv[2], sys.argv[3] +assert doc["status"] == "ok" +assert doc["source"]["actual_commit"] == sha_b +assert doc["source"]["commit_verified"] is True +wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") +assert wheel["actual_sha256"] == py_sha +assert wheel["actual_sha256"] != sha_b +assert wheel["digest_verified"] is True +assert wheel.get("built_from_commit") in (None, "") +assert "verified" not in wheel +assert doc["pairing"]["stack_paired_proven"] is False +assert doc["pairing"]["status"] == "unproven" +assert "MinimaxV2.5_EP2" in doc["cases_preserved"] +assert "GLM45Air_EP2" in doc["cases_preserved"] +print("ok-url receipt fields checked") +PY + [[ "$(git -C "${root}/ok-url/PaddleFleet" rev-parse HEAD)" == "${sha_b}" ]] + + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + PADDLEFLEET_BUILD_CMD="mkdir -p '${root}/ok-build/dist' && cp '${root}/art/py.whl' '${root}/ok-build/dist/paddlefleet-0.0.0-py3-none-any.whl'" \ + PADDLEFLEET_BUILD_OPS_CMD="mkdir -p '${root}/ok-build/dist' && cp '${root}/art/ops.whl' '${root}/ok-build/dist/paddlefleet_ops-0.0.0-py3-none-any.whl'" \ + bash "${script}" --dest "${root}/ok-build" + python3 - "${root}/ok-build/paddlefleet_alignment_pin_receipt.json" "${sha_b}" "${py_sha}" <<'PY' +import json, sys +doc = json.load(open(sys.argv[1])) +sha_b, py_sha = sys.argv[2], sys.argv[3] +assert doc["status"] == "ok" +wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") +assert wheel["actual_sha256"] == py_sha +assert wheel["actual_sha256"] != sha_b +assert wheel["origin"] == "build" +assert wheel["built_from_commit"] == sha_b +assert doc["source"]["actual_commit"] == sha_b +assert doc["pairing"]["stack_paired_proven"] is False +assert doc["pairing"]["status"] == "built_from_checked_out_pin" +print("ok-build receipt fields checked") +PY + grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-build/paddlefleet_alignment_pin.env" + + grep -q 'CodeSync/develop/PaddleFleet.tar' "${script}" + grep -q 'PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl' "${script}" + + echo "select_paddlefleet_alignment_pin self-test OK" +} + +if [[ "${RUN_SELF_TEST}" == 1 ]]; then + run_self_test + exit 0 +fi + +mkdir -p "${DEST}" +case "${MODE}" in + stack-paired) fetch_stack_paired ;; + develop) fetch_default ;; + *) fail "unknown ALIGNMENT_PADDLEFLEET_MODE=${MODE} (develop|stack-paired)" ;; +esac From 67dbededa4b65bbbc71dc005dce54d675bb314c2 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 18:09:30 +0800 Subject: [PATCH 17/33] Allow Node 20 on self-hosted citest checkout paddle-4 runner 2.337.0 forces Node 24 for actions/checkout@v3; node24 needs GLIBC 2.27/2.28 which this host lacks (job 101275701451). Keep checkout@v3 and opt into ACTIONS_ALLOW_USE_UNSECURE_NODE_VERSION. --- .github/workflows/citest.yaml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.github/workflows/citest.yaml b/.github/workflows/citest.yaml index 12776e60e3..f5ec87ddc1 100644 --- a/.github/workflows/citest.yaml +++ b/.github/workflows/citest.yaml @@ -41,6 +41,10 @@ jobs: # The type of runner that the job will run on runs-on: [self-hosted] timeout-minutes: 240 + env: + # self-hosted paddle-4 glibc < 2.27; runner forces Node 24 for + # actions/checkout@v3 (job 101275701451). Allow Node 20 so checkout runs. + ACTIONS_ALLOW_USE_UNSECURE_NODE_VERSION: true steps: - name: ResetFileMode shell: bash From af15c6972e2df7e33cfe45b14c23c4bbcdd87ceb Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 21:03:30 +0800 Subject: [PATCH 18/33] Pass stack-paired PaddleFleet source paths across docker exec steps Match Megatron #4: checkout COMMIT_ID on workflow_dispatch, export source trees from the pin selector, and consume that env in the alignment step so setup_venvs never sees the hardcoded 0.0.0 wheel. --- .../workflows/alignment_model_accuracy.yaml | 35 +++-- scripts/consume_paddlefleet_alignment_pin.sh | 143 ++++++++++++++++++ scripts/select_paddlefleet_alignment_pin.sh | 65 +++++++- scripts/test_paddlefleet_pin_handoff.sh | 78 ++++++++++ 4 files changed, 306 insertions(+), 15 deletions(-) create mode 100755 scripts/consume_paddlefleet_alignment_pin.sh create mode 100755 scripts/test_paddlefleet_pin_handoff.sh diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index b06448ec12..c24f82a9f0 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -122,17 +122,25 @@ jobs: git config --global --add safe.directory /workspace/ms-swift git pull git submodule update --init --recursive --force + git remote add upstream https://github.com/PFCCLab/ms-swift.git || true if [ -n "$PR_ID" ] && [ "$PR_ID" != "0" ]; then git fetch origin pull/${PR_ID}/head - git checkout -b PR_${PR_ID} FETCH_HEAD - git remote add upstream https://github.com/PFCCLab/ms-swift.git + git checkout -B PR_${PR_ID} FETCH_HEAD echo "Checking out ${BRANCH}..." - git fetch upstream ${BRANCH}:${BRANCH} - git merge ${BRANCH} --no-edit + git fetch upstream ${BRANCH}:${BRANCH} || true + git merge ${BRANCH} --no-edit || true git diff --numstat ${BRANCH} -- | awk "{print \$NF}" + elif [ -n "$COMMIT_ID" ]; then + echo "workflow_dispatch: checkout COMMIT_ID=${COMMIT_ID} (not BOS main)" + git fetch --all --tags || true + git fetch origin "$COMMIT_ID" || git fetch upstream "$COMMIT_ID" || true + git checkout --force "$COMMIT_ID" + git submodule update --init --recursive --force else - echo "Not in a pull_request event. Skipping PR-specific operations." + echo "Not in a pull_request event and COMMIT_ID empty. Leaving tarball HEAD." fi + echo "checked_out_head=$(git rev-parse HEAD)" + test -z "$COMMIT_ID" || test "$(git rev-parse HEAD)" = "$COMMIT_ID" git log --pretty=oneline -10 ' - name: Change python version @@ -163,8 +171,11 @@ jobs: if [ "${ALIGNMENT_PADDLEFLEET_MODE:-develop}" = "stack-paired" ]; then echo "::group::Select stack-paired PaddleFleet (fail-closed SHA)" python -m pip install uv + test -x /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh \ + || { echo "::error:: selector missing; checkout did not land COMMIT_ID"; exit 1; } bash /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh --dest /workspace cat /workspace/paddlefleet_alignment_pin_receipt.json + test -f /workspace/paddlefleet_alignment_pin.env echo "::endgroup::" echo "::group::Download remaining wheels from BOS" for url in \ @@ -205,12 +216,12 @@ jobs: source $work_dir/../../../proxy export PROXY_URL="${http_proxy}" - if [ -f /workspace/paddlefleet_alignment_pin.env ]; then - . /workspace/paddlefleet_alignment_pin.env - echo "loaded pin receipt ${PADDLEFLEET_PIN_RECEIPT:-/workspace/paddlefleet_alignment_pin_receipt.json}" - fi - export PADDLEFLEET_WHEEL_PATH="${PADDLEFLEET_WHEEL_PATH:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" - export PADDLEFLEET_OPS_WHEEL_PATH="${PADDLEFLEET_OPS_WHEEL_PATH:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" + bash /workspace/ms-swift/scripts/consume_paddlefleet_alignment_pin.sh \ + --env /workspace/paddlefleet_alignment_pin.env \ + --out /workspace/paddlefleet_alignment_pin.consumed.env + set -a + . /workspace/paddlefleet_alignment_pin.consumed.env + set +a export MEGATRON_CORE_WHEEL_PATH=/workspace/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl export MS_SWIFT_WHEEL_PATH=/workspace/upload/ms_swift-0.0.0-py3-none-any.whl export MCORE_BRIDGE_WHEEL_PATH=/workspace/mcore_bridge-0.0.0-py3-none-any.whl @@ -220,7 +231,7 @@ jobs: for whl in "$PADDLEFLEET_WHEEL_PATH" "$PADDLEFLEET_OPS_WHEEL_PATH" \ "$MS_SWIFT_WHEEL_PATH" "$MEGATRON_CORE_WHEEL_PATH" \ "$MCORE_BRIDGE_WHEEL_PATH"; do - [ -f "$whl" ] || [ -d "$whl" ] || { echo "::error:: missing wheel: $whl"; exit 1; } + [ -e "$whl" ] || { echo "::error:: missing wheel: $whl"; exit 1; } echo "using $whl" done python -m pip install uv diff --git a/scripts/consume_paddlefleet_alignment_pin.sh b/scripts/consume_paddlefleet_alignment_pin.sh new file mode 100755 index 0000000000..d7be1857c4 --- /dev/null +++ b/scripts/consume_paddlefleet_alignment_pin.sh @@ -0,0 +1,143 @@ +#!/usr/bin/env bash +# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. +# +# Consume selector output in a later docker exec. Source-mode paths must +# survive the step boundary; stack-paired must not fall back to the +# hardcoded 0.0.0 wheel filenames used by unpaired develop. + +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: consume_paddlefleet_alignment_pin.sh [--env FILE] [--out FILE] [--self-test] + +Reads paddlefleet_alignment_pin.env written by select_paddlefleet_alignment_pin.sh +and writes a consumed env file for setup_venvs.sh. Stack-paired refuses the +0.0.0 develop filename and requires existing file-or-directory paths. +EOF +} + +ENVFILE="${PADDLEFLEET_PIN_ENV:-/workspace/paddlefleet_alignment_pin.env}" +OUTFILE="${PADDLEFLEET_CONSUMED_ENV:-/workspace/paddlefleet_alignment_pin.consumed.env}" +RUN_SELF_TEST=0 +while [[ $# -gt 0 ]]; do + case "$1" in + --env) ENVFILE="${2:?}"; shift 2 ;; + --out) OUTFILE="${2:?}"; shift 2 ;; + --self-test) RUN_SELF_TEST=1; shift ;; + -h|--help) usage; exit 0 ;; + *) echo "unknown arg: $1" >&2; usage; exit 2 ;; + esac +done + +hardcoded_develop_wheel() { + case "$1" in + */paddlefleet-0.0.0-py3-none-linux_x86_64.whl) return 0 ;; + */paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl) return 0 ;; + *) return 1 ;; + esac +} + +write_consumed() { + mkdir -p "$(dirname "${OUTFILE}")" + cat >"${OUTFILE}" <&2 + echo "[paddlefleet-pin-consume] wheel=${PADDLEFLEET_WHEEL_PATH} ops=${PADDLEFLEET_OPS_WHEEL_PATH} mode=${ALIGNMENT_PADDLEFLEET_MODE} origin=${PADDLEFLEET_WHEEL_ORIGIN:-}" >&2 +} + +consume() { + local mode="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" + if [[ -f "${ENVFILE}" ]]; then + # shellcheck disable=SC1090 + set -a + # shellcheck disable=SC1090 + . "${ENVFILE}" + set +a + echo "[paddlefleet-pin-consume] loaded ${ENVFILE} receipt=${PADDLEFLEET_PIN_RECEIPT:-}" >&2 + mode="${ALIGNMENT_PADDLEFLEET_MODE:-${mode}}" + fi + ALIGNMENT_PADDLEFLEET_MODE="${mode}" + + if [[ "${mode}" == "stack-paired" ]]; then + [[ -f "${ENVFILE}" ]] || { + echo "::error:: stack-paired missing ${ENVFILE}; selector export did not cross docker exec" >&2 + exit 1 + } + [[ -n "${PADDLEFLEET_WHEEL_PATH:-}" && -n "${PADDLEFLEET_OPS_WHEEL_PATH:-}" ]] || { + echo "::error:: stack-paired env missing PADDLEFLEET_WHEEL_PATH or OPS path" >&2 + exit 1 + } + if hardcoded_develop_wheel "${PADDLEFLEET_WHEEL_PATH}" || hardcoded_develop_wheel "${PADDLEFLEET_OPS_WHEEL_PATH}"; then + echo "::error:: stack-paired refused hardcoded 0.0.0 wheel fallback: ${PADDLEFLEET_WHEEL_PATH} ${PADDLEFLEET_OPS_WHEEL_PATH}" >&2 + exit 1 + fi + else + PADDLEFLEET_WHEEL_PATH="${PADDLEFLEET_WHEEL_PATH:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" + PADDLEFLEET_OPS_WHEEL_PATH="${PADDLEFLEET_OPS_WHEEL_PATH:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" + fi + + if [[ ! -e "${PADDLEFLEET_WHEEL_PATH}" ]]; then + echo "::error:: missing paddlefleet path: ${PADDLEFLEET_WHEEL_PATH}" >&2 + exit 1 + fi + if [[ ! -e "${PADDLEFLEET_OPS_WHEEL_PATH}" ]]; then + echo "::error:: missing paddlefleet_ops path: ${PADDLEFLEET_OPS_WHEEL_PATH}" >&2 + exit 1 + fi + write_consumed +} + +run_self_test() { + local root self + self="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)/$(basename -- "${BASH_SOURCE[0]}")" + root="$(mktemp -d)" + trap 'rm -rf "${root}"' RETURN + mkdir -p "${root}/PaddleFleet/packages/paddlefleet_ops" + echo tree >"${root}/PaddleFleet/pyproject.toml" + echo ops >"${root}/PaddleFleet/packages/paddlefleet_ops/pyproject.toml" + + cat >"${root}/pin.env" <"${root}/missing.env" <&2 + exit 1 + fi + echo "consume_paddlefleet_alignment_pin self-test OK" +} + +if [[ "${RUN_SELF_TEST}" == 1 ]]; then + run_self_test + exit 0 +fi +consume diff --git a/scripts/select_paddlefleet_alignment_pin.sh b/scripts/select_paddlefleet_alignment_pin.sh index 71c2131318..2fdc7b2d06 100755 --- a/scripts/select_paddlefleet_alignment_pin.sh +++ b/scripts/select_paddlefleet_alignment_pin.sh @@ -121,6 +121,10 @@ pairing_fields() { if [[ "${MODE}" == "develop" ]]; then status="unpaired_default" reason="develop tarball and develop/latest wheels; not a stack pin" + elif [[ "${WHEEL_ORIGIN}" == "source_tree" && "${OPS_ORIGIN}" == "source_tree" \ + && "${SOURCE_VERIFIED}" == "true" ]]; then + status="source_tree_from_checked_out_pin" + reason="this invocation exported checked-out source trees; not a wheel digest proof" elif [[ "${WHEEL_ORIGIN}" == "build" && "${OPS_ORIGIN}" == "build" \ && "${WHEEL_BUILT_FROM_COMMIT}" == "${ACTUAL_SHA}" \ && "${OPS_BUILT_FROM_COMMIT}" == "${ACTUAL_SHA}" \ @@ -247,6 +251,9 @@ PADDLEFLEET_OPS_WHEEL_PATH=${PADDLEFLEET_OPS_WHEEL_PATH} ALIGNMENT_PADDLEFLEET_MODE=${MODE} PADDLEFLEET_PIN_RECEIPT=${DEST}/paddlefleet_alignment_pin_receipt.json PADDLEFLEET_SOURCE_COMMIT=${ACTUAL_SHA} +PADDLEFLEET_PIN_SHA=${PIN_SHA} +PADDLEFLEET_WHEEL_ORIGIN=${WHEEL_ORIGIN} +PADDLEFLEET_OPS_ORIGIN=${OPS_ORIGIN} EOF } @@ -413,11 +420,35 @@ fetch_default() { write_receipt "ok" "develop tarball and develop/latest wheels; unpaired with a stack pin" } +acquire_source_tree() { + local src="${DEST}/PaddleFleet" + local ops="${src}/packages/paddlefleet_ops" + [[ -d "${src}" ]] || fail "stack-paired source tree missing: ${src}" + [[ -f "${src}/pyproject.toml" ]] || fail "stack-paired source tree missing pyproject.toml: ${src}" + [[ -d "${ops}" ]] || fail "stack-paired ops source tree missing: ${ops}" + PADDLEFLEET_WHEEL_PATH="${src}" + PADDLEFLEET_OPS_WHEEL_PATH="${ops}" + WHEEL_ORIGIN="source_tree" + OPS_ORIGIN="source_tree" + WHEEL_BUILT_FROM_COMMIT="${ACTUAL_SHA}" + OPS_BUILT_FROM_COMMIT="${ACTUAL_SHA}" + WHEEL_DIGEST_VERIFIED=false + OPS_DIGEST_VERIFIED=false + LOADED_FROM="source_tree ${src} ${ops} from ${ACTUAL_SHA}" + log "source-tree paths ${src} ${ops}" +} + fetch_stack_paired() { log "mode=stack-paired" checkout_pin - acquire_wheel - acquire_ops + if [[ -n "${WHEEL_URL}" || -n "${OPS_URL}" || -n "${BUILD_CMD}" || -n "${BUILD_OPS_CMD}" ]]; then + acquire_wheel + acquire_ops + else + # No CI wheel URL and no explicit build: export the checked-out trees. + # A later docker exec must source paddlefleet_alignment_pin.env. + acquire_source_tree + fi export_receipt_env write_envfile local pair pairing_status @@ -473,7 +504,10 @@ run_self_test() { git -C "${root}/upstream" config user.email test@example.com git -C "${root}/upstream" config user.name test echo source-a >"${root}/upstream/README" - git -C "${root}/upstream" add README + mkdir -p "${root}/upstream/packages/paddlefleet_ops" + printf '%s\n' '[project]' 'name = "paddlefleet"' >"${root}/upstream/pyproject.toml" + printf '%s\n' '[project]' 'name = "paddlefleet-ops"' >"${root}/upstream/packages/paddlefleet_ops/pyproject.toml" + git -C "${root}/upstream" add README pyproject.toml packages git -C "${root}/upstream" commit -q -m a local sha_a sha_b sha_a="$(git -C "${root}/upstream" rev-parse HEAD)" @@ -619,6 +653,31 @@ print("ok-build receipt fields checked") PY grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-build/paddlefleet_alignment_pin.env" + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + bash "${script}" --dest "${root}/ok-source" + python3 - "${root}/ok-source/paddlefleet_alignment_pin_receipt.json" "${sha_b}" "${root}/ok-source" <<'PY' +import json, sys +doc = json.load(open(sys.argv[1])) +sha_b, dest = sys.argv[2], sys.argv[3] +assert doc["status"] == "ok" +assert doc["source"]["actual_commit"] == sha_b +assert doc["source"]["commit_verified"] is True +wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") +ops = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet_ops") +assert wheel["path"] == f"{dest}/PaddleFleet" +assert ops["path"] == f"{dest}/PaddleFleet/packages/paddlefleet_ops" +assert wheel["origin"] == "source_tree" +assert ops["origin"] == "source_tree" +assert doc["pairing"]["status"] == "source_tree_from_checked_out_pin" +assert doc["pairing"]["stack_paired_proven"] is False +print("ok-source receipt fields checked") +PY + grep -q "PADDLEFLEET_WHEEL_PATH=${root}/ok-source/PaddleFleet$" "${root}/ok-source/paddlefleet_alignment_pin.env" + grep -q "PADDLEFLEET_OPS_WHEEL_PATH=${root}/ok-source/PaddleFleet/packages/paddlefleet_ops$" "${root}/ok-source/paddlefleet_alignment_pin.env" + grep -q "PADDLEFLEET_WHEEL_ORIGIN=source_tree" "${root}/ok-source/paddlefleet_alignment_pin.env" + grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-source/paddlefleet_alignment_pin.env" + grep -q 'CodeSync/develop/PaddleFleet.tar' "${script}" grep -q 'PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl' "${script}" diff --git a/scripts/test_paddlefleet_pin_handoff.sh b/scripts/test_paddlefleet_pin_handoff.sh new file mode 100755 index 0000000000..948dcbe318 --- /dev/null +++ b/scripts/test_paddlefleet_pin_handoff.sh @@ -0,0 +1,78 @@ +#!/usr/bin/env bash +# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. +# +# End-to-end: selector source-mode -> new-shell consume -> setup_venvs consumer. +# This is the docker-exec step boundary. Isolated selector --self-test is not enough. + +set -euo pipefail + +ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" +SELECTOR="${ROOT}/select_paddlefleet_alignment_pin.sh" +CONSUME="${ROOT}/consume_paddlefleet_alignment_pin.sh" + +tmp="$(mktemp -d)" +trap 'rm -rf "${tmp}"' EXIT + +git init -q "${tmp}/upstream" +git -C "${tmp}/upstream" config user.email test@example.com +git -C "${tmp}/upstream" config user.name test +mkdir -p "${tmp}/upstream/packages/paddlefleet_ops" +printf '%s\n' '[project]' 'name = "paddlefleet"' >"${tmp}/upstream/pyproject.toml" +printf '%s\n' '[project]' 'name = "paddlefleet-ops"' >"${tmp}/upstream/packages/paddlefleet_ops/pyproject.toml" +echo src >"${tmp}/upstream/README" +git -C "${tmp}/upstream" add README pyproject.toml packages +git -C "${tmp}/upstream" commit -q -m pin +PIN="$(git -C "${tmp}/upstream" rev-parse HEAD)" + +# Step A: Get Whl equivalent (selector). +ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${PIN}" \ + PADDLEFLEET_GIT_URL="${tmp}/upstream" \ + bash "${SELECTOR}" --dest "${tmp}/ws" + +test -f "${tmp}/ws/paddlefleet_alignment_pin.env" +test -d "${tmp}/ws/PaddleFleet" + +# Step B: new docker exec — drop selector shell state, keep only files. +unset PADDLEFLEET_WHEEL_PATH PADDLEFLEET_OPS_WHEEL_PATH PADDLEFLEET_SOURCE_COMMIT || true +export ALIGNMENT_PADDLEFLEET_MODE=stack-paired +bash "${CONSUME}" --env "${tmp}/ws/paddlefleet_alignment_pin.env" --out "${tmp}/ws/consumed.env" + +# Step C: setup_venvs consumer. Fail if it would install the 0.0.0 develop wheel. +stub_setup="${tmp}/setup_venvs.sh" +cat >"${stub_setup}" <<'STUB' +#!/usr/bin/env bash +set -euo pipefail +# Mirrors setup_venvs.sh: it only sees exported PADDLEFLEET_WHEEL_PATH. +PADDLEFLEET_WHEEL="${PADDLEFLEET_WHEEL_PATH:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" +PADDLEFLEET_OPS_WHEEL="${PADDLEFLEET_OPS_WHEEL_PATH:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" +case "${PADDLEFLEET_WHEEL}" in + */paddlefleet-0.0.0-py3-none-linux_x86_64.whl) + echo "STUB_SETUP used hardcoded 0.0.0 wheel" >&2 + exit 1 + ;; +esac +[[ -d "${PADDLEFLEET_WHEEL}" || -f "${PADDLEFLEET_WHEEL}" ]] || { echo "missing ${PADDLEFLEET_WHEEL}" >&2; exit 1; } +[[ -d "${PADDLEFLEET_OPS_WHEEL}" || -f "${PADDLEFLEET_OPS_WHEEL}" ]] || { echo "missing ${PADDLEFLEET_OPS_WHEEL}" >&2; exit 1; } +echo "STUB_SETUP paddlefleet=${PADDLEFLEET_WHEEL}" +echo "STUB_SETUP ops=${PADDLEFLEET_OPS_WHEEL}" +STUB +chmod +x "${stub_setup}" + +set -a +# shellcheck disable=SC1090 +. "${tmp}/ws/consumed.env" +set +a +bash "${stub_setup}" | tee "${tmp}/setup.out" +grep -q "STUB_SETUP paddlefleet=${tmp}/ws/PaddleFleet" "${tmp}/setup.out" +grep -q "packages/paddlefleet_ops" "${tmp}/setup.out" + +# Negative: alignment step that ignores consumed.env and defaults 0.0.0 must fail the stub. +if PADDLEFLEET_WHEEL_PATH=/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl \ + PADDLEFLEET_OPS_WHEEL_PATH=/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl \ + bash "${stub_setup}"; then + echo "handoff FAIL: stub accepted 0.0.0 default" >&2 + exit 1 +fi + +echo "paddlefleet pin handoff OK pin=${PIN}" From dd542ca8beaa659f307861e69b5b50cf88c68460 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 21:08:56 +0800 Subject: [PATCH 19/33] Keep requested stack-paired mode/pin when consuming selector env Match Megatron #4: caller mode/pin stay authoritative; unproven develop fallback is refused; verified 0.0.0 names remain legal. --- .../workflows/alignment_model_accuracy.yaml | 2 + scripts/consume_paddlefleet_alignment_pin.sh | 294 +++++++++++++++--- scripts/select_paddlefleet_alignment_pin.sh | 3 + scripts/test_paddlefleet_pin_handoff.sh | 62 ++-- 4 files changed, 283 insertions(+), 78 deletions(-) diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index c24f82a9f0..1b87a09787 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -216,6 +216,8 @@ jobs: source $work_dir/../../../proxy export PROXY_URL="${http_proxy}" + ALIGNMENT_PADDLEFLEET_MODE="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" \ + PADDLEFLEET_PIN_SHA="${PADDLEFLEET_PIN_SHA:-}" \ bash /workspace/ms-swift/scripts/consume_paddlefleet_alignment_pin.sh \ --env /workspace/paddlefleet_alignment_pin.env \ --out /workspace/paddlefleet_alignment_pin.consumed.env diff --git a/scripts/consume_paddlefleet_alignment_pin.sh b/scripts/consume_paddlefleet_alignment_pin.sh index d7be1857c4..f9181d03ba 100755 --- a/scripts/consume_paddlefleet_alignment_pin.sh +++ b/scripts/consume_paddlefleet_alignment_pin.sh @@ -1,9 +1,11 @@ #!/usr/bin/env bash # Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. # -# Consume selector output in a later docker exec. Source-mode paths must -# survive the step boundary; stack-paired must not fall back to the -# hardcoded 0.0.0 wheel filenames used by unpaired develop. +# Consume selector output in a later docker exec. The caller mode/pin stay +# authoritative: a leftover develop env must not silently downgrade +# stack-paired. A 0.0.0 filename is allowed when the receipt proves an +# explicit URL+sha256 (or a source tree / in-invocation build). Unproven +# develop/latest fallback is refused. set -euo pipefail @@ -11,9 +13,11 @@ usage() { cat <<'EOF' Usage: consume_paddlefleet_alignment_pin.sh [--env FILE] [--out FILE] [--self-test] -Reads paddlefleet_alignment_pin.env written by select_paddlefleet_alignment_pin.sh -and writes a consumed env file for setup_venvs.sh. Stack-paired refuses the -0.0.0 develop filename and requires existing file-or-directory paths. +Reads paddlefleet_alignment_pin.env + receipt from select_paddlefleet_alignment_pin.sh +and writes a consumed env file for setup_venvs.sh. + +Caller ALIGNMENT_PADDLEFLEET_MODE / PADDLEFLEET_PIN_SHA are the request. +The env file must match that request; it does not override them. EOF } @@ -30,12 +34,101 @@ while [[ $# -gt 0 ]]; do esac done -hardcoded_develop_wheel() { - case "$1" in - */paddlefleet-0.0.0-py3-none-linux_x86_64.whl) return 0 ;; - */paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl) return 0 ;; - *) return 1 ;; +fail() { + echo "::error:: $*" >&2 + exit 1 +} + +parse_envfile() { + FILE_MODE="" + FILE_PIN="" + FILE_SOURCE="" + FILE_RECEIPT="" + FILE_WHEEL="" + FILE_OPS="" + FILE_WHEEL_ORIGIN="" + FILE_OPS_ORIGIN="" + FILE_WHEEL_DIGEST="" + FILE_OPS_DIGEST="" + [[ -f "${ENVFILE}" ]] || return 0 + local line k v + while IFS= read -r line || [[ -n "${line}" ]]; do + [[ -z "${line}" || "${line}" == \#* ]] && continue + k="${line%%=*}" + v="${line#*=}" + case "${k}" in + ALIGNMENT_PADDLEFLEET_MODE) FILE_MODE="${v}" ;; + PADDLEFLEET_PIN_SHA) FILE_PIN="${v}" ;; + PADDLEFLEET_SOURCE_COMMIT) FILE_SOURCE="${v}" ;; + PADDLEFLEET_PIN_RECEIPT) FILE_RECEIPT="${v}" ;; + PADDLEFLEET_WHEEL_PATH) FILE_WHEEL="${v}" ;; + PADDLEFLEET_OPS_WHEEL_PATH) FILE_OPS="${v}" ;; + PADDLEFLEET_WHEEL_ORIGIN) FILE_WHEEL_ORIGIN="${v}" ;; + PADDLEFLEET_OPS_ORIGIN) FILE_OPS_ORIGIN="${v}" ;; + PADDLEFLEET_WHEEL_DIGEST_VERIFIED) FILE_WHEEL_DIGEST="${v}" ;; + PADDLEFLEET_OPS_DIGEST_VERIFIED) FILE_OPS_DIGEST="${v}" ;; + esac + done <"${ENVFILE}" +} + +# Unproven develop fallback: develop_latest origin, or a default 0.0.0 +# filename with no digest proof. A verified ci_metadata/build artifact may +# legally keep the 0.0.0 filename. +unproven_develop_fallback() { + local path="$1" origin="$2" digest="$3" + case "${origin}" in + develop_latest) return 0 ;; + source_tree|build|ci_metadata) + [[ "${origin}" == develop_latest ]] && return 0 + return 1 + ;; + esac + case "${path}" in + */paddlefleet-0.0.0-py3-none-linux_x86_64.whl|*/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl) + [[ "${digest}" == "true" ]] && return 1 + return 0 + ;; esac + return 1 +} + +check_receipt() { + local receipt="$1" requested_mode="$2" requested_pin="$3" + [[ -f "${receipt}" ]] || fail "stack-paired missing receipt ${receipt}" + python3 - "${receipt}" "${requested_mode}" "${requested_pin}" <<'PY' +import json, sys +path, requested_mode, requested_pin = sys.argv[1:4] +doc = json.load(open(path, encoding="utf-8")) +if doc.get("status") != "ok": + raise SystemExit(f"receipt status={doc.get('status')!r} is not ok") +if requested_mode == "stack-paired": + if doc.get("mode") != "stack-paired": + raise SystemExit( + f"requested stack-paired but receipt mode={doc.get('mode')!r}" + ) + src = doc.get("source") or {} + if requested_pin: + exp = src.get("expected_commit") or "" + act = src.get("actual_commit") or "" + if requested_pin not in (exp, act): + raise SystemExit( + f"receipt source pin mismatch requested={requested_pin} " + f"expected={exp} actual={act}" + ) + if src.get("commit_verified") is not True: + raise SystemExit("receipt source commit_verified is not true") + pairing = (doc.get("pairing") or {}).get("status") + if pairing == "unpaired_default": + raise SystemExit("receipt pairing.status=unpaired_default") + for art in doc.get("artifacts") or []: + origin = art.get("origin") or "" + url = art.get("url") or "" + if origin == "develop_latest": + raise SystemExit(f"artifact {art.get('name')} origin=develop_latest") + if "/develop/latest/" in url or "CodeSync/develop/" in url: + raise SystemExit(f"artifact {art.get('name')} unpaired develop URL") +print("receipt matches request") +PY } write_consumed() { @@ -46,55 +139,97 @@ PADDLEFLEET_OPS_WHEEL_PATH=${PADDLEFLEET_OPS_WHEEL_PATH} ALIGNMENT_PADDLEFLEET_MODE=${ALIGNMENT_PADDLEFLEET_MODE} PADDLEFLEET_PIN_RECEIPT=${PADDLEFLEET_PIN_RECEIPT:-} PADDLEFLEET_SOURCE_COMMIT=${PADDLEFLEET_SOURCE_COMMIT:-} +PADDLEFLEET_PIN_SHA=${PADDLEFLEET_PIN_SHA:-} PADDLEFLEET_WHEEL_ORIGIN=${PADDLEFLEET_WHEEL_ORIGIN:-} PADDLEFLEET_OPS_ORIGIN=${PADDLEFLEET_OPS_ORIGIN:-} EOF echo "[paddlefleet-pin-consume] wrote ${OUTFILE}" >&2 - echo "[paddlefleet-pin-consume] wheel=${PADDLEFLEET_WHEEL_PATH} ops=${PADDLEFLEET_OPS_WHEEL_PATH} mode=${ALIGNMENT_PADDLEFLEET_MODE} origin=${PADDLEFLEET_WHEEL_ORIGIN:-}" >&2 + echo "[paddlefleet-pin-consume] requested_mode=${ALIGNMENT_PADDLEFLEET_MODE} pin=${PADDLEFLEET_PIN_SHA:-} wheel=${PADDLEFLEET_WHEEL_PATH} origin=${PADDLEFLEET_WHEEL_ORIGIN:-}" >&2 } consume() { - local mode="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" - if [[ -f "${ENVFILE}" ]]; then - # shellcheck disable=SC1090 - set -a - # shellcheck disable=SC1090 - . "${ENVFILE}" - set +a - echo "[paddlefleet-pin-consume] loaded ${ENVFILE} receipt=${PADDLEFLEET_PIN_RECEIPT:-}" >&2 - mode="${ALIGNMENT_PADDLEFLEET_MODE:-${mode}}" - fi - ALIGNMENT_PADDLEFLEET_MODE="${mode}" - - if [[ "${mode}" == "stack-paired" ]]; then - [[ -f "${ENVFILE}" ]] || { - echo "::error:: stack-paired missing ${ENVFILE}; selector export did not cross docker exec" >&2 - exit 1 - } - [[ -n "${PADDLEFLEET_WHEEL_PATH:-}" && -n "${PADDLEFLEET_OPS_WHEEL_PATH:-}" ]] || { - echo "::error:: stack-paired env missing PADDLEFLEET_WHEEL_PATH or OPS path" >&2 - exit 1 - } - if hardcoded_develop_wheel "${PADDLEFLEET_WHEEL_PATH}" || hardcoded_develop_wheel "${PADDLEFLEET_OPS_WHEEL_PATH}"; then - echo "::error:: stack-paired refused hardcoded 0.0.0 wheel fallback: ${PADDLEFLEET_WHEEL_PATH} ${PADDLEFLEET_OPS_WHEEL_PATH}" >&2 - exit 1 + local requested_mode="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" + local requested_pin="${PADDLEFLEET_PIN_SHA:-}" + + parse_envfile + + if [[ "${requested_mode}" == "stack-paired" ]]; then + [[ -f "${ENVFILE}" ]] || fail "stack-paired missing ${ENVFILE}; selector export did not cross docker exec" + [[ "${FILE_MODE}" == "stack-paired" ]] || fail "requested stack-paired but env mode=${FILE_MODE:-empty} (will not consume a develop leftover)" + if [[ -n "${requested_pin}" ]]; then + local file_id="${FILE_PIN:-${FILE_SOURCE}}" + [[ "${file_id}" == "${requested_pin}" ]] || fail "requested pin ${requested_pin} != env pin/source ${file_id:-empty}" + fi + local receipt="${FILE_RECEIPT:-}" + if [[ -z "${receipt}" && -f "${ENVFILE%/*}/paddlefleet_alignment_pin_receipt.json" ]]; then + receipt="${ENVFILE%/*}/paddlefleet_alignment_pin_receipt.json" + fi + check_receipt "${receipt}" "${requested_mode}" "${requested_pin}" + PADDLEFLEET_WHEEL_PATH="${FILE_WHEEL}" + PADDLEFLEET_OPS_WHEEL_PATH="${FILE_OPS}" + PADDLEFLEET_WHEEL_ORIGIN="${FILE_WHEEL_ORIGIN}" + PADDLEFLEET_OPS_ORIGIN="${FILE_OPS_ORIGIN}" + PADDLEFLEET_PIN_RECEIPT="${receipt}" + PADDLEFLEET_SOURCE_COMMIT="${FILE_SOURCE}" + PADDLEFLEET_PIN_SHA="${requested_pin:-${FILE_PIN}}" + ALIGNMENT_PADDLEFLEET_MODE="stack-paired" + [[ -n "${PADDLEFLEET_WHEEL_PATH}" && -n "${PADDLEFLEET_OPS_WHEEL_PATH}" ]] \ + || fail "stack-paired env missing PADDLEFLEET_WHEEL_PATH or OPS path" + if unproven_develop_fallback "${PADDLEFLEET_WHEEL_PATH}" "${FILE_WHEEL_ORIGIN}" "${FILE_WHEEL_DIGEST}"; then + fail "stack-paired refused unproven develop fallback for paddlefleet: path=${PADDLEFLEET_WHEEL_PATH} origin=${FILE_WHEEL_ORIGIN:-empty} digest_verified=${FILE_WHEEL_DIGEST:-false}" + fi + if unproven_develop_fallback "${PADDLEFLEET_OPS_WHEEL_PATH}" "${FILE_OPS_ORIGIN}" "${FILE_OPS_DIGEST}"; then + fail "stack-paired refused unproven develop fallback for paddlefleet_ops: path=${PADDLEFLEET_OPS_WHEEL_PATH} origin=${FILE_OPS_ORIGIN:-empty} digest_verified=${FILE_OPS_DIGEST:-false}" fi else - PADDLEFLEET_WHEEL_PATH="${PADDLEFLEET_WHEEL_PATH:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" - PADDLEFLEET_OPS_WHEEL_PATH="${PADDLEFLEET_OPS_WHEEL_PATH:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" + ALIGNMENT_PADDLEFLEET_MODE="develop" + PADDLEFLEET_WHEEL_PATH="${FILE_WHEEL:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" + PADDLEFLEET_OPS_WHEEL_PATH="${FILE_OPS:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" + PADDLEFLEET_WHEEL_ORIGIN="${FILE_WHEEL_ORIGIN:-develop_latest}" + PADDLEFLEET_OPS_ORIGIN="${FILE_OPS_ORIGIN:-develop_latest}" + PADDLEFLEET_PIN_RECEIPT="${FILE_RECEIPT:-}" + PADDLEFLEET_SOURCE_COMMIT="${FILE_SOURCE:-}" + PADDLEFLEET_PIN_SHA="${requested_pin}" fi if [[ ! -e "${PADDLEFLEET_WHEEL_PATH}" ]]; then - echo "::error:: missing paddlefleet path: ${PADDLEFLEET_WHEEL_PATH}" >&2 - exit 1 + fail "missing paddlefleet path: ${PADDLEFLEET_WHEEL_PATH}" fi if [[ ! -e "${PADDLEFLEET_OPS_WHEEL_PATH}" ]]; then - echo "::error:: missing paddlefleet_ops path: ${PADDLEFLEET_OPS_WHEEL_PATH}" >&2 - exit 1 + fail "missing paddlefleet_ops path: ${PADDLEFLEET_OPS_WHEEL_PATH}" fi write_consumed } +write_min_receipt() { + local path="$1" mode="$2" pin="$3" wheel="$4" ops="$5" origin="$6" digest="$7" pairing="$8" + python3 - "${path}" "${mode}" "${pin}" "${wheel}" "${ops}" "${origin}" "${digest}" "${pairing}" <<'PY' +import json, sys +path, mode, pin, wheel, ops, origin, digest, pairing = sys.argv[1:9] +digest_ok = digest == "true" +doc = { + "schema": "paddlefleet-alignment-pin/v1", + "status": "ok", + "mode": mode, + "source": { + "expected_commit": pin or None, + "actual_commit": pin or None, + "commit_verified": bool(pin) and mode == "stack-paired", + }, + "artifacts": [ + {"name": "paddlefleet", "path": wheel, "url": None, "origin": origin, + "digest_verified": digest_ok, "expected_sha256": "abc" if digest_ok else None, + "actual_sha256": "abc" if digest_ok else None}, + {"name": "paddlefleet_ops", "path": ops, "url": None, "origin": origin, + "digest_verified": digest_ok, "expected_sha256": "def" if digest_ok else None, + "actual_sha256": "def" if digest_ok else None}, + ], + "pairing": {"stack_paired_proven": False, "status": pairing, "reason": "fixture"}, +} +open(path, "w", encoding="utf-8").write(json.dumps(doc, indent=2) + "\n") +PY +} + run_self_test() { local root self self="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)/$(basename -- "${BASH_SOURCE[0]}")" @@ -103,36 +238,93 @@ run_self_test() { mkdir -p "${root}/PaddleFleet/packages/paddlefleet_ops" echo tree >"${root}/PaddleFleet/pyproject.toml" echo ops >"${root}/PaddleFleet/packages/paddlefleet_ops/pyproject.toml" + local pin="aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + write_min_receipt "${root}/receipt.json" stack-paired "${pin}" \ + "${root}/PaddleFleet" "${root}/PaddleFleet/packages/paddlefleet_ops" \ + source_tree false source_tree_from_checked_out_pin cat >"${root}/pin.env" <"${root}/missing.env" <"${root}/devwhl/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" + echo dummy >"${root}/devwhl/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" + write_min_receipt "${root}/develop-receipt.json" develop "" \ + "${root}/devwhl/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" \ + "${root}/devwhl/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" \ + develop_latest false unpaired_default + cat >"${root}/develop.env" <&2 + exit 1 + fi + + # 2) verified URL+hash artifact may keep the 0.0.0 filename. + write_min_receipt "${root}/named-receipt.json" stack-paired "${pin}" \ + "${root}/devwhl/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" \ + "${root}/devwhl/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" \ + ci_metadata true unproven + cat >"${root}/named.env" <&2 + ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${pin}" \ + bash "${self}" --env "${root}/named.env" --out "${root}/named.consumed.env" + grep -q "paddlefleet-0.0.0-py3-none-linux_x86_64.whl" "${root}/named.consumed.env" + + # Unproven 0.0.0 fallback (no origin, no digest) still fails. + cat >"${root}/bare.env" <&2 exit 1 fi + echo "consume_paddlefleet_alignment_pin self-test OK" } diff --git a/scripts/select_paddlefleet_alignment_pin.sh b/scripts/select_paddlefleet_alignment_pin.sh index 2fdc7b2d06..55e5f7f362 100755 --- a/scripts/select_paddlefleet_alignment_pin.sh +++ b/scripts/select_paddlefleet_alignment_pin.sh @@ -254,6 +254,8 @@ PADDLEFLEET_SOURCE_COMMIT=${ACTUAL_SHA} PADDLEFLEET_PIN_SHA=${PIN_SHA} PADDLEFLEET_WHEEL_ORIGIN=${WHEEL_ORIGIN} PADDLEFLEET_OPS_ORIGIN=${OPS_ORIGIN} +PADDLEFLEET_WHEEL_DIGEST_VERIFIED=${WHEEL_DIGEST_VERIFIED} +PADDLEFLEET_OPS_DIGEST_VERIFIED=${OPS_DIGEST_VERIFIED} EOF } @@ -676,6 +678,7 @@ PY grep -q "PADDLEFLEET_WHEEL_PATH=${root}/ok-source/PaddleFleet$" "${root}/ok-source/paddlefleet_alignment_pin.env" grep -q "PADDLEFLEET_OPS_WHEEL_PATH=${root}/ok-source/PaddleFleet/packages/paddlefleet_ops$" "${root}/ok-source/paddlefleet_alignment_pin.env" grep -q "PADDLEFLEET_WHEEL_ORIGIN=source_tree" "${root}/ok-source/paddlefleet_alignment_pin.env" + grep -q "PADDLEFLEET_WHEEL_DIGEST_VERIFIED=false" "${root}/ok-source/paddlefleet_alignment_pin.env" grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-source/paddlefleet_alignment_pin.env" grep -q 'CodeSync/develop/PaddleFleet.tar' "${script}" diff --git a/scripts/test_paddlefleet_pin_handoff.sh b/scripts/test_paddlefleet_pin_handoff.sh index 948dcbe318..990d9924f8 100755 --- a/scripts/test_paddlefleet_pin_handoff.sh +++ b/scripts/test_paddlefleet_pin_handoff.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash # Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. # -# End-to-end: selector source-mode -> new-shell consume -> setup_venvs consumer. -# This is the docker-exec step boundary. Isolated selector --self-test is not enough. +# End-to-end path check: selector source-mode -> new-shell consume -> +# setup_venvs *path* consumer. This proves docker-exec handoff of source +# paths. It is NOT a uv install / real setup_venvs / numerical CI run. +# Isolated selector --self-test is not enough for the step boundary. set -euo pipefail @@ -34,28 +36,41 @@ test -f "${tmp}/ws/paddlefleet_alignment_pin.env" test -d "${tmp}/ws/PaddleFleet" # Step B: new docker exec — drop selector shell state, keep only files. +# Requested mode/pin stay on the caller; leftover develop env must not win. unset PADDLEFLEET_WHEEL_PATH PADDLEFLEET_OPS_WHEEL_PATH PADDLEFLEET_SOURCE_COMMIT || true -export ALIGNMENT_PADDLEFLEET_MODE=stack-paired -bash "${CONSUME}" --env "${tmp}/ws/paddlefleet_alignment_pin.env" --out "${tmp}/ws/consumed.env" +ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${PIN}" \ + bash "${CONSUME}" --env "${tmp}/ws/paddlefleet_alignment_pin.env" --out "${tmp}/ws/consumed.env" -# Step C: setup_venvs consumer. Fail if it would install the 0.0.0 develop wheel. -stub_setup="${tmp}/setup_venvs.sh" +# Negative: leftover develop env + requested stack-paired must fail closed. +mkdir -p "${tmp}/dev" +echo dummy >"${tmp}/dev/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" +echo dummy >"${tmp}/dev/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" +cat >"${tmp}/dev.env" <&2 + exit 1 +fi + +# Step C: path consumer only. Does not run uv or setup_venvs.sh. +stub_setup="${tmp}/setup_path_consumer.sh" cat >"${stub_setup}" <<'STUB' #!/usr/bin/env bash set -euo pipefail -# Mirrors setup_venvs.sh: it only sees exported PADDLEFLEET_WHEEL_PATH. -PADDLEFLEET_WHEEL="${PADDLEFLEET_WHEEL_PATH:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" -PADDLEFLEET_OPS_WHEEL="${PADDLEFLEET_OPS_WHEEL_PATH:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" -case "${PADDLEFLEET_WHEEL}" in - */paddlefleet-0.0.0-py3-none-linux_x86_64.whl) - echo "STUB_SETUP used hardcoded 0.0.0 wheel" >&2 - exit 1 - ;; -esac +# Mirrors setup_venvs.sh reading PADDLEFLEET_WHEEL_PATH. Path presence only. +PADDLEFLEET_WHEEL="${PADDLEFLEET_WHEEL_PATH:?missing PADDLEFLEET_WHEEL_PATH}" +PADDLEFLEET_OPS_WHEEL="${PADDLEFLEET_OPS_WHEEL_PATH:?missing PADDLEFLEET_OPS_WHEEL_PATH}" [[ -d "${PADDLEFLEET_WHEEL}" || -f "${PADDLEFLEET_WHEEL}" ]] || { echo "missing ${PADDLEFLEET_WHEEL}" >&2; exit 1; } [[ -d "${PADDLEFLEET_OPS_WHEEL}" || -f "${PADDLEFLEET_OPS_WHEEL}" ]] || { echo "missing ${PADDLEFLEET_OPS_WHEEL}" >&2; exit 1; } -echo "STUB_SETUP paddlefleet=${PADDLEFLEET_WHEEL}" -echo "STUB_SETUP ops=${PADDLEFLEET_OPS_WHEEL}" +echo "PATH_CONSUMER paddlefleet=${PADDLEFLEET_WHEEL}" +echo "PATH_CONSUMER ops=${PADDLEFLEET_OPS_WHEEL}" +echo "PATH_CONSUMER not_uv_install=true" STUB chmod +x "${stub_setup}" @@ -64,15 +79,8 @@ set -a . "${tmp}/ws/consumed.env" set +a bash "${stub_setup}" | tee "${tmp}/setup.out" -grep -q "STUB_SETUP paddlefleet=${tmp}/ws/PaddleFleet" "${tmp}/setup.out" +grep -q "PATH_CONSUMER paddlefleet=${tmp}/ws/PaddleFleet" "${tmp}/setup.out" grep -q "packages/paddlefleet_ops" "${tmp}/setup.out" +grep -q "not_uv_install=true" "${tmp}/setup.out" -# Negative: alignment step that ignores consumed.env and defaults 0.0.0 must fail the stub. -if PADDLEFLEET_WHEEL_PATH=/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl \ - PADDLEFLEET_OPS_WHEEL_PATH=/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl \ - bash "${stub_setup}"; then - echo "handoff FAIL: stub accepted 0.0.0 default" >&2 - exit 1 -fi - -echo "paddlefleet pin handoff OK pin=${PIN}" +echo "paddlefleet pin handoff PATH_PASS pin=${PIN} (not uv install, not CI)" From 4d7c4186e035e0e09ac832f8a273aa69dd33d274 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sat, 5 Sep 2026 23:02:04 +0800 Subject: [PATCH 20/33] Stop Get Whl after selector failure and drop nested quotes Selector clone failure must fail the docker exec (set -e) instead of continuing wget/build. Receipt checks live in a helper so the single-quoted -c script has no nested quotes. Missing pin.env after an error receipt is selector failure, not env-handoff. --- .../workflows/alignment_model_accuracy.yaml | 9 +- scripts/consume_paddlefleet_alignment_pin.sh | 37 +++- scripts/require_paddlefleet_selector_ok.sh | 40 +++++ scripts/test_alignment_workflow_shell.sh | 161 ++++++++++++++++++ scripts/test_paddlefleet_pin_handoff.sh | 28 +++ 5 files changed, 271 insertions(+), 4 deletions(-) create mode 100755 scripts/require_paddlefleet_selector_ok.sh create mode 100755 scripts/test_alignment_workflow_shell.sh diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index 1b87a09787..73d0af7d6d 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -161,6 +161,7 @@ jobs: - name: Get Whl run: | docker exec -t $container_name /bin/bash -c ' + set -eo pipefail . /opt/conda/etc/profile.d/conda.sh conda activate py_$python_version python --version @@ -173,9 +174,10 @@ jobs: python -m pip install uv test -x /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh \ || { echo "::error:: selector missing; checkout did not land COMMIT_ID"; exit 1; } - bash /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh --dest /workspace - cat /workspace/paddlefleet_alignment_pin_receipt.json - test -f /workspace/paddlefleet_alignment_pin.env + bash /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh --dest /workspace \ + || { echo "::error:: selector failed; stop Get Whl (do not download remaining wheels or build ms-swift)"; exit 1; } + bash /workspace/ms-swift/scripts/require_paddlefleet_selector_ok.sh /workspace \ + || { echo "::error:: selector receipt status is not ok; stop Get Whl"; exit 1; } echo "::endgroup::" echo "::group::Download remaining wheels from BOS" for url in \ @@ -209,6 +211,7 @@ jobs: - name: alignment_model_accuracy run: | docker exec -t $container_name /bin/bash -c ' + set -eo pipefail . /opt/conda/etc/profile.d/conda.sh conda activate py_$python_version python --version diff --git a/scripts/consume_paddlefleet_alignment_pin.sh b/scripts/consume_paddlefleet_alignment_pin.sh index f9181d03ba..1ba0b8e56b 100755 --- a/scripts/consume_paddlefleet_alignment_pin.sh +++ b/scripts/consume_paddlefleet_alignment_pin.sh @@ -154,7 +154,24 @@ consume() { parse_envfile if [[ "${requested_mode}" == "stack-paired" ]]; then - [[ -f "${ENVFILE}" ]] || fail "stack-paired missing ${ENVFILE}; selector export did not cross docker exec" + if [[ ! -f "${ENVFILE}" ]]; then + local rec="${ENVFILE%/*}/paddlefleet_alignment_pin_receipt.json" + local rec_status="" + if [[ -f "${rec}" ]]; then + rec_status="$(python3 - "${rec}" <<'PY' +import json, sys +print(json.load(open(sys.argv[1])).get("status") or "") +PY +)" + fi + if [[ "${rec_status}" == "error" ]]; then + fail "stack-paired missing ${ENVFILE} because selector failed (error receipt ${rec}); Get Whl must exit on selector failure. This is not proof that a generated env failed to cross docker exec" + fi + if [[ "${rec_status}" == "ok" ]]; then + fail "stack-paired missing ${ENVFILE}; selector wrote ok receipt but env did not cross docker exec" + fi + fail "stack-paired missing ${ENVFILE} and no selector receipt" + fi [[ "${FILE_MODE}" == "stack-paired" ]] || fail "requested stack-paired but env mode=${FILE_MODE:-empty} (will not consume a develop leftover)" if [[ -n "${requested_pin}" ]]; then local file_id="${FILE_PIN:-${FILE_SOURCE}}" @@ -325,6 +342,24 @@ EOF exit 1 fi + # Selector clone/fail: error receipt, no env. Missing env is the + # consequence, not proof a generated env failed to cross docker exec. + mkdir -p "${root}/sel-fail" + cat >"${root}/sel-fail/paddlefleet_alignment_pin_receipt.json" <<'EOF' +{"schema":"paddlefleet-alignment-pin/v1","status":"error","detail":"git clone failed: github.com:443","mode":"stack-paired"} +EOF + if ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${pin}" \ + bash "${self}" --env "${root}/sel-fail/paddlefleet_alignment_pin.env" \ + --out "${root}/sel-fail.consumed.env" 2>"${root}/sel-fail.err"; then + echo "self-test FAIL: selector error receipt was consumed" >&2 + exit 1 + fi + grep -q "because selector failed" "${root}/sel-fail.err" + if grep -q "selector wrote ok receipt but env did not cross docker exec" "${root}/sel-fail.err"; then + echo "self-test FAIL: selector error misclassified as env-handoff" >&2 + exit 1 + fi + echo "consume_paddlefleet_alignment_pin self-test OK" } diff --git a/scripts/require_paddlefleet_selector_ok.sh b/scripts/require_paddlefleet_selector_ok.sh new file mode 100755 index 0000000000..f34221204a --- /dev/null +++ b/scripts/require_paddlefleet_selector_ok.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. +# +# Called from Get Whl after select_paddlefleet_alignment_pin.sh. +# Keep this file free of nested quotes so the workflow docker exec +# single-quoted -c script can invoke it without breaking the host shell. +# A selector error receipt is not an env-handoff failure. + +set -euo pipefail + +DEST="${1:-/workspace}" +REC="${DEST}/paddlefleet_alignment_pin_receipt.json" +ENVF="${DEST}/paddlefleet_alignment_pin.env" + +if [[ ! -f "${REC}" ]]; then + echo "::error:: missing selector receipt ${REC}" >&2 + exit 1 +fi + +python3 - "${REC}" <<'PY' +import json +import sys + +path = sys.argv[1] +doc = json.load(open(path, encoding="utf-8")) +status = doc.get("status") +if status != "ok": + raise SystemExit( + f"selector receipt status={status!r} is not ok; " + "Get Whl must stop (do not download remaining wheels or build)" + ) +print("selector receipt status=ok") +PY + +if [[ ! -f "${ENVF}" ]]; then + echo "::error:: selector did not write ${ENVF}" >&2 + exit 1 +fi + +cat "${REC}" diff --git a/scripts/test_alignment_workflow_shell.sh b/scripts/test_alignment_workflow_shell.sh new file mode 100755 index 0000000000..196e32c4ad --- /dev/null +++ b/scripts/test_alignment_workflow_shell.sh @@ -0,0 +1,161 @@ +#!/usr/bin/env bash +# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. +# +# Syntax-check every workflow `run:` block, then execute the extracted +# Get Whl docker-exec body against a failing selector. Independent +# helper --self-test is not this check. + +set -euo pipefail + +ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" +YAML="${ROOT}/../.github/workflows/alignment_model_accuracy.yaml" +SELECTOR_PATH="/workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh" +REQUIRE_PATH="/workspace/ms-swift/scripts/require_paddlefleet_selector_ok.sh" +BUILD_MARKER="build ms-swift" + +python3 - "${YAML}" "${ROOT}" "${SELECTOR_PATH}" "${REQUIRE_PATH}" "${BUILD_MARKER}" <<'PY' +import os, re, subprocess, sys, tempfile, textwrap, pathlib, json, stat + +yaml_path, scripts_root, selector_path, require_path, build_marker = sys.argv[1:6] +text = pathlib.Path(yaml_path).read_text() +lines = text.splitlines(True) + +# Extract literal `run: |` blocks by indent. +blocks = [] +i = 0 +while i < len(lines): + m = re.match(r"^(\s*)run:\s*\|\s*$", lines[i]) + if not m: + i += 1 + continue + indent = len(m.group(1)) + name = "unnamed" + for j in range(i, -1, -1): + nm = re.match(r"^\s+- name:\s*(.*)$", lines[j]) + if nm: + name = nm.group(1).strip() + break + i += 1 + body = [] + while i < len(lines): + line = lines[i] + if line.strip() == "": + body.append(line) + i += 1 + continue + lead = len(line) - len(line.lstrip(" ")) + if lead <= indent and line.strip(): + break + body.append(line[indent + 2 :] if lead >= indent + 2 else line.lstrip()) + i += 1 + blocks.append((name, "".join(body))) + +if not blocks: + raise SystemExit(f"no run: | blocks in {yaml_path}") + +tmp = pathlib.Path(tempfile.mkdtemp(prefix="yaml-run-")) +print(f"extracted {len(blocks)} run blocks from {yaml_path}") +for n, body in blocks: + p = tmp / (re.sub(r"[^A-Za-z0-9._-]+", "_", n) + ".sh") + p.write_text("#!/usr/bin/env bash\n" + body) + r = subprocess.run(["bash", "-n", str(p)], capture_output=True, text=True) + if r.returncode != 0: + raise SystemExit(f"bash -n FAIL {n}: {r.stderr}") + print(f"bash -n OK run:{n}") + + # Reconstruct docker exec -c quoting: the host sees + # docker exec ... /bin/bash -c 'INNER' + # INNER must not contain an unescaped single quote. + for m in re.finditer(r"""/bin/bash -c\s+'""", body): + start = m.end() + end = body.find("'\n", start) + if end < 0: + end = body.rfind("'") + inner = body[start:end] + if "'" in inner: + raise SystemExit( + f"nested single quote inside docker exec -c in {n}: " + f"{inner[inner.find(chr(39))-40:inner.find(chr(39))+40]!r}" + ) + inner_p = tmp / (p.stem + ".docker-inner.sh") + inner_p.write_text("#!/usr/bin/env bash\n" + inner + "\n") + r = subprocess.run(["bash", "-n", str(inner_p)], capture_output=True, text=True) + if r.returncode != 0: + raise SystemExit(f"bash -n FAIL docker-inner {n}: {r.stderr}") + print(f"bash -n OK docker-inner:{n} (no nested single quotes)") + +getwhl = next((b for n, b in blocks if n == "Get Whl"), None) +if getwhl is None: + raise SystemExit("Get Whl run block missing") +m = re.search(r"""/bin/bash -c\s+'""", getwhl) +if not m: + raise SystemExit("Get Whl docker exec -c missing") +inner = getwhl[m.end():] +end = inner.rfind("'") +inner = inner[:end] + +# Fixture: run the extracted Get Whl body with a failing selector. +ws = tmp / "ws" +(ws / "ms-swift/scripts").mkdir(parents=True) +(ws / "upload").mkdir(parents=True) +selector = ws / "ms-swift/scripts/select_paddlefleet_alignment_pin.sh" +require_src = pathlib.Path(scripts_root) / "require_paddlefleet_selector_ok.sh" +require_dst = ws / "ms-swift/scripts/require_paddlefleet_selector_ok.sh" +require_dst.write_text(require_src.read_text()) +require_dst.chmod(require_dst.stat().st_mode | stat.S_IXUSR) +selector.write_text(textwrap.dedent("""\ + #!/usr/bin/env bash + set -euo pipefail + dest="${2:-/workspace}" + mkdir -p "${dest}" + cat >"${dest}/paddlefleet_alignment_pin_receipt.json" <<'EOF' + {"schema":"paddlefleet-alignment-pin/v1","status":"error","detail":"git clone failed: github.com:443","mode":"stack-paired"} + EOF + echo "[paddlefleet-pin] FAIL: git clone failed: github.com:443" >&2 + echo "::error:: git clone failed: github.com:443" >&2 + exit 1 + """)) +selector.chmod(selector.stat().st_mode | stat.S_IXUSR) +(ws / "ms-swift/scripts/dependence").mkdir(parents=True, exist_ok=True) +(ws / "ms-swift/scripts/dependence/build.sh").write_text("#!/usr/bin/env bash\necho BUILD_RAN > /workspace/upload/BUILD_RAN\n") +(ws / "ms-swift/scripts/dependence/build.sh").chmod(0o755) + +# Rewrite extracted inner to use the fixture workspace and stub tools. +rewritten = inner +rewritten = rewritten.replace("/workspace", str(ws)) +rewritten = rewritten.replace("conda activate py_$python_version", "true") +rewritten = rewritten.replace(". /opt/conda/etc/profile.d/conda.sh", "true") +rewritten = "#!/usr/bin/env bash\nexport ALIGNMENT_PADDLEFLEET_MODE=stack-paired\nexport python_version=3.12\n" + rewritten +# Stub wget / pip / python / ldconfig so a leaked continue would be visible. +bin = tmp / "bin" +bin.mkdir() +(bin / "wget").write_text("#!/usr/bin/env bash\necho WGET_RAN \"$@\" >> '%s/WGET_RAN'\nexit 0\n" % ws) +(bin / "python").write_text("#!/usr/bin/env bash\necho PY_RAN \"$@\" >> '%s/PY_RAN'\nexit 0\n" % ws) +(bin / "python3").write_text("#!/usr/bin/env bash\nexec /usr/bin/python3 \"$@\"\n") +(bin / "pip").write_text("#!/usr/bin/env bash\necho PIP_RAN \"$@\" >> '%s/PIP_RAN'\nexit 0\n" % ws) +(bin / "ldconfig").write_text("#!/usr/bin/env bash\nexit 0\n") +for f in bin.iterdir(): + f.chmod(0o755) + +script = tmp / "getwhl.extracted.sh" +script.write_text(rewritten) +script.chmod(0o755) +env = os.environ.copy() +env["PATH"] = str(bin) + ":" + env.get("PATH", "") +env["ALIGNMENT_PADDLEFLEET_MODE"] = "stack-paired" +r = subprocess.run(["bash", str(script)], capture_output=True, text=True, env=env) +log = (r.stdout or "") + (r.stderr or "") +print("extracted Get Whl rc=", r.returncode) +print(log[-2000:]) +if r.returncode == 0: + raise SystemExit("FAIL: extracted Get Whl continued after selector failure") +if (ws / "WGET_RAN").exists(): + raise SystemExit("FAIL: wget ran after selector failure") +if (ws / "upload/BUILD_RAN").exists(): + raise SystemExit("FAIL: build.sh ran after selector failure") +if "selector failed; stop Get Whl" not in log and "git clone failed" not in log: + raise SystemExit("FAIL: extracted Get Whl did not surface selector failure") +print("extracted Get Whl fixture: selector fail stopped remaining wheels and build") +print("workflow shell checks OK") +PY +echo "alignment workflow shell PATH_PASS (extracted YAML, not helper-only)" diff --git a/scripts/test_paddlefleet_pin_handoff.sh b/scripts/test_paddlefleet_pin_handoff.sh index 990d9924f8..72cc719998 100755 --- a/scripts/test_paddlefleet_pin_handoff.sh +++ b/scripts/test_paddlefleet_pin_handoff.sh @@ -58,6 +58,34 @@ if ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${PIN}" \ exit 1 fi +# Negative: selector clone/fail writes error receipt and no env. Consume +# must name selector failure. Missing env here is the consequence, not +# proof that a generated env failed to cross docker exec. +REQUIRE="${ROOT}/require_paddlefleet_selector_ok.sh" +mkdir -p "${tmp}/sel-fail" +cat >"${tmp}/sel-fail/paddlefleet_alignment_pin_receipt.json" <<'EOF' +{"schema":"paddlefleet-alignment-pin/v1","status":"error","detail":"git clone failed: github.com:443","mode":"stack-paired"} +EOF +if bash "${REQUIRE}" "${tmp}/sel-fail" >"${tmp}/sel-fail.require.out" 2>"${tmp}/sel-fail.require.err"; then + echo "handoff FAIL: require_ok accepted error receipt" >&2 + exit 1 +fi +grep -q "status='error' is not ok" "${tmp}/sel-fail.require.err" \ + || grep -q 'status="error" is not ok' "${tmp}/sel-fail.require.err" \ + || grep -q "status=error is not ok" "${tmp}/sel-fail.require.err" \ + || grep -q "selector receipt status=" "${tmp}/sel-fail.require.err" +if ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${PIN}" \ + bash "${CONSUME}" --env "${tmp}/sel-fail/paddlefleet_alignment_pin.env" \ + --out "${tmp}/sel-fail.consumed.env" 2>"${tmp}/sel-fail.err"; then + echo "handoff FAIL: selector error receipt was consumed" >&2 + exit 1 +fi +grep -q "because selector failed" "${tmp}/sel-fail.err" +if grep -q "selector wrote ok receipt but env did not cross docker exec" "${tmp}/sel-fail.err"; then + echo "handoff FAIL: selector error misclassified as env-handoff" >&2 + exit 1 +fi + # Step C: path consumer only. Does not run uv or setup_venvs.sh. stub_setup="${tmp}/setup_path_consumer.sh" cat >"${stub_setup}" <<'STUB' From b4631b5a85fa2dab48f53a0eb62b2ce99302de49 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Sun, 6 Sep 2026 22:57:29 +0800 Subject: [PATCH 21/33] fix: fetch PaddleFleet pin at depth 1 with bounded retries stack-paired checkout_pin used a full default-branch git clone with no retry. Swift 34038640242 failed Get Whl on curl 56 / early EOF before ops, so Fleet 09bb4bd4 was not evaluated. Fetch origin $PIN_SHA at --depth=1 --no-tags, detach-checkout, and keep exact HEAD. Transient RPC/curl-56 retries up to 3 clean dests; HTTP 401/403, missing SHA, and missing repo fail closed on attempt 1 with commit_verified=false. --- scripts/select_paddlefleet_alignment_pin.sh | 231 ++++++++++++++++++-- 1 file changed, 210 insertions(+), 21 deletions(-) diff --git a/scripts/select_paddlefleet_alignment_pin.sh b/scripts/select_paddlefleet_alignment_pin.sh index 55e5f7f362..cfa778fa9d 100755 --- a/scripts/select_paddlefleet_alignment_pin.sh +++ b/scripts/select_paddlefleet_alignment_pin.sh @@ -20,7 +20,9 @@ # Cases are not filtered. # # Explicit (ALIGNMENT_PADDLEFLEET_MODE=stack-paired): fail-closed. -# Checkout PADDLEFLEET_PIN_SHA; git rev-parse HEAD must equal the pin. +# Fetch PADDLEFLEET_PIN_SHA at depth 1 (not a full default-branch clone). +# Transient git RPC/curl failures retry up to 3 clean dests; other +# fetch errors fail closed. git rev-parse HEAD must equal the pin. # Artifacts: caller URL+sha256 (from Build Fleet whl / Actions metadata) # or build from the checked-out tree. Independent digest matches do not # prove the wheels were produced from that commit — receipt records @@ -292,26 +294,65 @@ require_digest() { fi } +# Auth / missing-object / missing-repo are permanent. Do not treat a +# generic "RPC failed" as transient: HTTP 401/403 also say RPC failed. +_is_permanent_git_err() { + case "$1" in + *"HTTP 401"*|*"HTTP 403"*|*"Authentication failed"*|*"access denied"*|*"Access denied"*|*"Permission denied"*|*"not our ref"*|*"does not appear to be a git repository"*|*"Could not read from remote repository"*|*"remote: Write access"*) + return 0 + ;; + esac + return 1 +} + +_is_transient_git_err() { + _is_permanent_git_err "$1" && return 1 + case "$1" in + *"curl 56"*|*"Connection timed out"*|*"Couldn't connect to server"*|*"Failed to connect to github.com port 443"*|*"early EOF"*|*"fetch-pack: unexpected disconnect"*|*"bytes of body are still expected"*|*"invalid index-pack output"*|*"Connection reset by peer"*) + return 0 + ;; + esac + return 1 +} + checkout_pin() { [[ "${PIN_SHA}" =~ ^[0-9a-fA-F]{40}$ ]] || fail "stack-paired requires PADDLEFLEET_PIN_SHA (40 hex), got '${PIN_SHA}'" PIN_SHA="$(printf '%s' "${PIN_SHA}" | tr 'A-F' 'a-f')" - rm -rf "${DEST}/PaddleFleet" - log "clone ${GIT_URL}" - if ! git clone --quiet "${GIT_URL}" "${DEST}/PaddleFleet" >/dev/null 2>"${DEST}/.git-clone.err"; then - fail "git clone failed: $(tr '\n' ' ' <"${DEST}/.git-clone.err")" - fi - git -C "${DEST}/PaddleFleet" config advice.detachedHead false || true - log "checkout ${PIN_SHA}" - if ! git -C "${DEST}/PaddleFleet" checkout --quiet --force "${PIN_SHA}" >/dev/null 2>"${DEST}/.git-co.err"; then - fail "git checkout failed for ${PIN_SHA}: $(tr '\n' ' ' <"${DEST}/.git-co.err")" - fi - ACTUAL_SHA="$(git -C "${DEST}/PaddleFleet" rev-parse HEAD)" - if [[ "${ACTUAL_SHA}" != "${PIN_SHA}" ]]; then - SOURCE_VERIFIED=false - fail "stack-paired source SHA mismatch expected=${PIN_SHA} actual=${ACTUAL_SHA}" - fi - SOURCE_VERIFIED=true - log "source commit verified ${ACTUAL_SHA}" + local dest="${DEST}/PaddleFleet" + local attempt max_attempts=3 + local last_err="" + local fetch_err="${DEST}/.git-fetch.err" + for attempt in $(seq 1 "${max_attempts}"); do + rm -rf "${dest}" + log "fetch ${GIT_URL} ${PIN_SHA} (attempt ${attempt}/${max_attempts})" + if ! git init --quiet "${dest}" >/dev/null 2>"${DEST}/.git-init.err"; then + fail "git init failed: $(tr '\n' ' ' <"${DEST}/.git-init.err")" + fi + if ! git -C "${dest}" remote add origin "${GIT_URL}" >/dev/null 2>"${DEST}/.git-remote.err"; then + fail "git remote add failed: $(tr '\n' ' ' <"${DEST}/.git-remote.err")" + fi + git -C "${dest}" config advice.detachedHead false || true + if GIT_TERMINAL_PROMPT=0 git -C "${dest}" fetch --depth=1 --no-tags origin "${PIN_SHA}" \ + >/dev/null 2>"${fetch_err}"; then + if ! git -C "${dest}" checkout --quiet --force --detach "${PIN_SHA}" >/dev/null 2>"${DEST}/.git-co.err"; then + fail "git checkout failed for ${PIN_SHA}: $(tr '\n' ' ' <"${DEST}/.git-co.err")" + fi + ACTUAL_SHA="$(git -C "${dest}" rev-parse HEAD)" + if [[ "${ACTUAL_SHA}" != "${PIN_SHA}" ]]; then + SOURCE_VERIFIED=false + fail "stack-paired source SHA mismatch expected=${PIN_SHA} actual=${ACTUAL_SHA}" + fi + SOURCE_VERIFIED=true + log "source commit verified ${ACTUAL_SHA}" + return 0 + fi + last_err="$(tr '\n' ' ' <"${fetch_err}")" + if ! _is_transient_git_err "${last_err}"; then + fail "git fetch failed: ${last_err}" + fi + log "transient fetch failure attempt ${attempt}/${max_attempts}: ${last_err}" + done + fail "git fetch failed: ${last_err}" } # Sets DEST_PATH and DEST_SHA in the caller. Must run in this shell so @@ -557,7 +598,7 @@ run_self_test() { PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ bash "${script}" --dest "${root}/m2" - expect_fail "${root}/m3" "git checkout failed" \ + expect_fail "${root}/m3" "git fetch failed" \ run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ PADDLEFLEET_PIN_SHA="0000000000000000000000000000000000000000" \ PADDLEFLEET_GIT_URL="${root}/upstream" \ @@ -599,12 +640,159 @@ PY PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ bash "${script}" --dest "${root}/m6" - expect_fail "${root}/m7" "git clone failed" \ + expect_fail "${root}/m7" "git fetch failed" \ run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ PADDLEFLEET_PIN_SHA="${sha_b}" \ PADDLEFLEET_GIT_URL="${root}/no-such-remote" \ bash "${script}" --dest "${root}/m7" + assert_error_unverified() { + python3 - "$1" <<'PY' +import json, sys +doc = json.load(open(sys.argv[1])) +assert doc["status"] == "error", doc +assert doc["source"]["commit_verified"] is False, doc["source"] +print("error receipt commit_verified=false") +PY + } + + assert_pin_checkout() { + local repo="$1" sha="$2" + [[ "$(git -C "${repo}" rev-parse HEAD)" == "${sha}" ]] + [[ -f "${repo}/.git/shallow" ]] || { echo "self-test FAIL: missing ${repo}/.git/shallow" >&2; exit 1; } + [[ "$(git -C "${repo}" rev-list --count HEAD)" == 1 ]] || { + echo "self-test FAIL: ${repo} rev-list count != 1 (not depth=1)" >&2 + exit 1 + } + if git -C "${repo}" symbolic-ref -q HEAD >/dev/null; then + echo "self-test FAIL: ${repo} HEAD is a branch, not detached pin" >&2 + git -C "${repo}" symbolic-ref HEAD >&2 + exit 1 + fi + } + + write_git_wrapper() { + local bindir="$1" mode="$2" + local real_git + real_git="$(command -v git)" + mkdir -p "${bindir}" + cat >"${bindir}/git" <>"\$STALE" + exit 99 + fi + if [[ "\$has_depth" -ne 1 || "\$has_no_tags" -ne 1 ]]; then + echo "fetch missing --depth=1/--no-tags: \${args[*]}" >>"\$STALE" + exit 99 + fi + count=\$((count + 1)) + printf '%s\n' "\$count" >"\$STATE" + case "${mode}" in + 401) + echo "RPC failed; HTTP 401 curl 22 The requested URL returned error: 401" >&2 + echo "fatal: Authentication failed" >&2 + printf 'sentinel\n' >"\$workdir/.retry-sentinel" + exit 128 + ;; + exhaust) + echo "error: RPC failed; curl 56 Recv failure: Connection timed out" >&2 + echo "error: 9515 bytes of body are still expected" >&2 + echo "fatal: early EOF" >&2 + printf 'sentinel\n' >"\$workdir/.retry-sentinel" + exit 128 + ;; + retry) + if [[ "\$count" -lt 3 ]]; then + echo "error: RPC failed; curl 56 Recv failure: Connection timed out" >&2 + echo "error: 9515 bytes of body are still expected" >&2 + echo "fetch-pack: unexpected disconnect while reading sideband packet" >&2 + echo "fatal: early EOF" >&2 + echo "fatal: fetch-pack: invalid index-pack output" >&2 + printf 'sentinel\n' >"\$workdir/.retry-sentinel" + exit 128 + fi + ;; + esac +fi +exec "\$real" "\$@" +GITWRAP + chmod +x "${bindir}/git" + } + + assert_error_unverified "${root}/m3/paddlefleet_alignment_pin_receipt.json" + assert_error_unverified "${root}/m7/paddlefleet_alignment_pin_receipt.json" + + # Permanent HTTP 401 (also says RPC failed) must not retry. + write_git_wrapper "${root}/bin-401" 401 + expect_fail "${root}/m8" "git fetch failed" \ + env PATH="${root}/bin-401:${root}/bin:${PATH}" \ + ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + bash "${script}" --dest "${root}/m8" + [[ "$(cat "${root}/bin-401/count")" == 1 ]] + [[ ! -f "${root}/bin-401/stale" ]] + assert_error_unverified "${root}/m8/paddlefleet_alignment_pin_receipt.json" + + # First two fetches emit Swift 34038640242 curl-56 / early-EOF and leave + # a dest sentinel; the next fetch must see a clean dest. Third fetch is + # real git. Depth=1 and detached pin, not a default-branch clone. + write_git_wrapper "${root}/bin-retry" retry + run PATH="${root}/bin-retry:${root}/bin:${PATH}" \ + ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + bash "${script}" --dest "${root}/ok-retry" + [[ "$(cat "${root}/bin-retry/count")" == 3 ]] + [[ ! -f "${root}/bin-retry/stale" ]] + [[ ! -e "${root}/ok-retry/PaddleFleet/.retry-sentinel" ]] + assert_pin_checkout "${root}/ok-retry/PaddleFleet" "${sha_b}" + python3 - "${root}/ok-retry/paddlefleet_alignment_pin_receipt.json" "${sha_b}" <<'PY' +import json, sys +doc = json.load(open(sys.argv[1])) +assert doc["status"] == "ok" +assert doc["source"]["actual_commit"] == sys.argv[2] +assert doc["source"]["commit_verified"] is True +print("ok-retry receipt exact HEAD verified") +PY + + # Exhausted transient retries: wrapper count is 3, dest cleaned between + # attempts, error receipt stays unverified. + write_git_wrapper "${root}/bin-fail" exhaust + expect_fail "${root}/m9" "git fetch failed" \ + env PATH="${root}/bin-fail:${root}/bin:${PATH}" \ + ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ + PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ + bash "${script}" --dest "${root}/m9" + [[ "$(cat "${root}/bin-fail/count")" == 3 ]] + [[ ! -f "${root}/bin-fail/stale" ]] + assert_error_unverified "${root}/m9/paddlefleet_alignment_pin_receipt.json" + run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ PADDLEFLEET_WHEEL_URL="${root}/art/py.whl" \ @@ -631,7 +819,7 @@ assert "MinimaxV2.5_EP2" in doc["cases_preserved"] assert "GLM45Air_EP2" in doc["cases_preserved"] print("ok-url receipt fields checked") PY - [[ "$(git -C "${root}/ok-url/PaddleFleet" rev-parse HEAD)" == "${sha_b}" ]] + assert_pin_checkout "${root}/ok-url/PaddleFleet" "${sha_b}" run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ @@ -680,6 +868,7 @@ PY grep -q "PADDLEFLEET_WHEEL_ORIGIN=source_tree" "${root}/ok-source/paddlefleet_alignment_pin.env" grep -q "PADDLEFLEET_WHEEL_DIGEST_VERIFIED=false" "${root}/ok-source/paddlefleet_alignment_pin.env" grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-source/paddlefleet_alignment_pin.env" + assert_pin_checkout "${root}/ok-source/PaddleFleet" "${sha_b}" grep -q 'CodeSync/develop/PaddleFleet.tar' "${script}" grep -q 'PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl' "${script}" From 8876e57005f7469d2d776481d6285c8802594a2b Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Mon, 7 Sep 2026 14:59:14 +0800 Subject: [PATCH 22/33] ci: bound and log alignment checkout operations --- .../workflows/alignment_model_accuracy.yaml | 47 +++++++++++---- scripts/test_checkout_observability.sh | 60 +++++++++++++++++++ 2 files changed, 94 insertions(+), 13 deletions(-) create mode 100755 scripts/test_checkout_observability.sh diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index 73d0af7d6d..c9941a10ea 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -104,38 +104,59 @@ jobs: - name: Checkout Code run: | docker exec -t $container_name /bin/bash -c ' + set -eo pipefail rm -rf * .[^.]* source $work_dir/../../../proxy + command -v timeout >/dev/null || { echo "checkout requires timeout" >&2; exit 127; } + # Keep this inline: the requested repository has not been checked out yet. + # Log labels, never command arguments (which can contain credentials). + checkout_step() { + local label="$1" limit="$2" status=0 started=$SECONDS + shift 2 + echo "checkout begin: ${label} utc=$(date -u +%FT%TZ) timeout=${limit}" + timeout --signal=TERM --kill-after=30s "$limit" "$@" || status=$? + echo "checkout end: ${label} status=${status} elapsed=$((SECONDS - started))s" + if [ "$status" -eq 124 ]; then + echo "checkout timed out: ${label} limit=${limit} status=${status}" >&2 + elif [ "$status" -eq 137 ]; then + echo "checkout terminated: ${label} status=137 (timeout escalation or SIGKILL)" >&2 + elif [ "$status" -ne 0 ]; then + echo "checkout failed: ${label} status=${status}" >&2 + fi + return "$status" + } if [ "${ALIGNMENT_PADDLEFLEET_MODE:-develop}" = "stack-paired" ]; then echo "stack-paired: skip CodeSync/develop PaddleFleet.tar; selector runs after python" else echo "Download PaddleFleet form https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" - wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate + checkout_step paddlefleet-download 600s wget -q --no-proxy --timeout=60 --tries=3 https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar - cd PaddleFleet && git pull && cd - + cd PaddleFleet + checkout_step paddlefleet-pull 300s git pull + cd - fi echo "Download ms-swift form https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz" - wget -q --no-proxy https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate + checkout_step swift-download 600s wget -q --no-proxy --timeout=60 --tries=3 https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate rm -rf ms-swift && tar zxf ms-swift.tar.gz && rm -rf ms-swift.tar.gz cd ms-swift git config --global --add safe.directory /workspace/ms-swift - git pull - git submodule update --init --recursive --force + checkout_step swift-pull 300s git pull + checkout_step swift-initial-submodules 900s git submodule update --init --recursive --force git remote add upstream https://github.com/PFCCLab/ms-swift.git || true if [ -n "$PR_ID" ] && [ "$PR_ID" != "0" ]; then - git fetch origin pull/${PR_ID}/head - git checkout -B PR_${PR_ID} FETCH_HEAD + checkout_step swift-pr-fetch 300s git fetch origin pull/${PR_ID}/head + checkout_step swift-pr-checkout 120s git checkout -B PR_${PR_ID} FETCH_HEAD echo "Checking out ${BRANCH}..." - git fetch upstream ${BRANCH}:${BRANCH} || true - git merge ${BRANCH} --no-edit || true + checkout_step swift-base-fetch 300s git fetch upstream ${BRANCH}:${BRANCH} || true + checkout_step swift-base-merge 120s git merge ${BRANCH} --no-edit || true git diff --numstat ${BRANCH} -- | awk "{print \$NF}" elif [ -n "$COMMIT_ID" ]; then echo "workflow_dispatch: checkout COMMIT_ID=${COMMIT_ID} (not BOS main)" - git fetch --all --tags || true - git fetch origin "$COMMIT_ID" || git fetch upstream "$COMMIT_ID" || true - git checkout --force "$COMMIT_ID" - git submodule update --init --recursive --force + checkout_step swift-fetch-all 900s git fetch --all --tags || true + checkout_step swift-commit-fetch-origin 300s git fetch origin "$COMMIT_ID" || checkout_step swift-commit-fetch-upstream 300s git fetch upstream "$COMMIT_ID" || true + checkout_step swift-commit-checkout 120s git checkout --force "$COMMIT_ID" + checkout_step swift-target-submodules 900s git submodule update --init --recursive --force else echo "Not in a pull_request event and COMMIT_ID empty. Leaving tarball HEAD." fi diff --git a/scripts/test_checkout_observability.sh b/scripts/test_checkout_observability.sh new file mode 100755 index 0000000000..6d954a273f --- /dev/null +++ b/scripts/test_checkout_observability.sh @@ -0,0 +1,60 @@ +#!/usr/bin/env bash +# Exercise the checkout wrapper extracted from the actual workflow, without network access. +set -euo pipefail +ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" +PYTHON_BIN="${PYTHON_BIN:-python3}" +"$PYTHON_BIN" - "$ROOT/.github/workflows/alignment_model_accuracy.yaml" <<'PYTEST' +import pathlib +import re +import shlex +import subprocess +import sys +import tempfile +import textwrap +import time + +text = pathlib.Path(sys.argv[1]).read_text() +checkout = text.split(" - name: Checkout Code\n", 1)[1].split(" - name:", 1)[0] +match = re.search(r"^ checkout_step\(\) \{\n.*?^ \}\n", checkout, re.M | re.S) +assert match, "actual checkout wrapper missing" +helper = textwrap.dedent(match.group()) +subprocess.run(["bash", "-n"], input=helper, text=True, check=True) +assert "set -eo pipefail" in checkout +assert not re.search(r"^\s*(?:wget |git (?:pull|fetch|submodule) )", checkout, re.M), "unbounded network command" +assert 'test "$(git rev-parse HEAD)" = "$COMMIT_ID"' in checkout, "exact HEAD check removed" + +secret = "fixture-credential-must-not-be-logged" +def run(label, limit, command): + invocation = shlex.join(["checkout_step", label, limit, *command, secret]) + script = helper + "\nset -e\n" + invocation + "\necho reached-next-step\n" + started = time.monotonic() + result = subprocess.run(["bash"], input=script, text=True, capture_output=True, timeout=8) + elapsed = time.monotonic() - started + output = result.stdout + result.stderr + assert secret not in output, output + assert "checkout begin: " + label in output, output + assert "checkout end: " + label in output, output + return result.returncode, output, elapsed + +rc, output, _ = run("success", "2s", ["bash", "-c", "exit 0"]) +assert rc == 0 and "status=0" in output and "reached-next-step" in output, output +print("PASS: success logs begin/end and continues without printing arguments") + +rc, output, _ = run("failure", "2s", ["bash", "-c", "exit 7"]) +assert rc == 7 and "checkout failed: failure status=7" in output and "reached-next-step" not in output, output +print("PASS: command failure preserves exit status and stops checkout") + +with tempfile.TemporaryDirectory(prefix="checkout-timeout-") as directory: + marker = pathlib.Path(directory) / "marker" + command = 'printf started > "$1"; sleep 20; printf finished > "$1"' + rc, output, elapsed = run("timeout", "0.2s", ["bash", "-c", command, "fixture", str(marker)]) + assert marker.read_text() == "started", output + assert rc == 124 and elapsed < 5, (rc, elapsed, output) + assert "checkout timed out: timeout" in output and "reached-next-step" not in output, output +print("PASS: actual timeout ends a running command and prevents continuation") + +rc, output, _ = run("killed", "2s", ["bash", "-c", "exit 137"]) +assert rc == 137 and "checkout terminated: killed" in output and "checkout timed out:" not in output, output +print("PASS: exit 137 is not falsely identified as a proven timeout") +print("All checkout observability fixtures passed") +PYTEST From c3a5c2afee5577a899c2cf977394f222432d817d Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Mon, 7 Sep 2026 16:02:40 +0800 Subject: [PATCH 23/33] ci: disable interactive pager in alignment checkout --- .../workflows/alignment_model_accuracy.yaml | 2 +- scripts/test_checkout_no_pager.sh | 65 +++++++++++++++++++ 2 files changed, 66 insertions(+), 1 deletion(-) create mode 100755 scripts/test_checkout_no_pager.sh diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index c9941a10ea..cbf4d60d6b 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -162,7 +162,7 @@ jobs: fi echo "checked_out_head=$(git rev-parse HEAD)" test -z "$COMMIT_ID" || test "$(git rev-parse HEAD)" = "$COMMIT_ID" - git log --pretty=oneline -10 + checkout_step swift-history 30s git --no-pager log --pretty=oneline -10 ' - name: Change python version run: | diff --git a/scripts/test_checkout_no_pager.sh b/scripts/test_checkout_no_pager.sh new file mode 100755 index 0000000000..21a1335b65 --- /dev/null +++ b/scripts/test_checkout_no_pager.sh @@ -0,0 +1,65 @@ +#!/usr/bin/env bash +# Reproduce an interactive Git pager in a real PTY, then check the workflow command. +set -euo pipefail +ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" +"${PYTHON_BIN:-python3}" - "$ROOT" <<'PYTEST' +import fcntl +import os +import pathlib +import pty +import select +import shlex +import shutil +import signal +import struct +import subprocess +import sys +import termios +import time + +root = pathlib.Path(sys.argv[1]) +text = (root / ".github/workflows/alignment_model_accuracy.yaml").read_text() +line = next(line.strip() for line in text.splitlines() if "checkout_step swift-history " in line) +actual = shlex.split(line)[3:] +assert actual == ["git", "--no-pager", "log", "--pretty=oneline", "-10"], actual +assert shutil.which("less"), "PTY regression fixture requires less" + +def observe(command): + master, slave = pty.openpty() + fcntl.ioctl(slave, termios.TIOCSWINSZ, struct.pack("HHHH", 8, 60, 0, 0)) + env = dict(os.environ, TERM="xterm", GIT_PAGER="less", LESS="-FRX") + proc = subprocess.Popen(command, cwd=root, env=env, stdin=slave, stdout=slave, + stderr=slave, start_new_session=True) + os.close(slave) + output = bytearray() + try: + until = time.monotonic() + 2 + while time.monotonic() < until and proc.poll() is None: + if select.select([master], [], [], 0.05)[0]: + try: + output.extend(os.read(master, 65536)) + except OSError: + break + # PTY EOF can arrive just before the child is reaped. + try: + proc.wait(timeout=0.2) + needed_input = False + except subprocess.TimeoutExpired: + needed_input = True + if needed_input: + os.write(master, b"q") + result = proc.wait(timeout=3) + assert result == 0, (result, bytes(output)[-200:]) + assert output, "expected Git history output" + return needed_input + finally: + if proc.poll() is None: + os.killpg(proc.pid, signal.SIGKILL) + proc.wait() + os.close(master) + +assert observe(["git", "log", "--pretty=oneline", "-10"]), "old command should await pager input in PTY" +print("PASS: old git log waits for input in PTY and exits on q") +assert not observe(actual), "workflow Git history must exit without input" +print("PASS: actual workflow --no-pager command exits without input in same PTY") +PYTEST From 58a02a0b927c07bfc82866789b99f202bded48db Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Tue, 8 Sep 2026 21:58:32 +0800 Subject: [PATCH 24/33] fix(megatron): preserve native accuracy loss and TP1 graph behavior Signed-off-by: Zhan Rongrui --- swift/megatron/callbacks/print.py | 156 ++++++++++++++++++ swift/megatron/init.py | 39 ++++- swift/megatron/trainers/base.py | 10 +- swift/megatron/trainers/trainer.py | 7 +- tests/megatron/test_accuracy_bridge_tp1.py | 123 ++++++++++++++ tests/megatron/test_accuracy_loss_and_norm.py | 89 ++++++++++ .../test_model_repro_machine_outputs.py | 74 +++++++++ 7 files changed, 486 insertions(+), 12 deletions(-) create mode 100644 tests/megatron/test_accuracy_bridge_tp1.py create mode 100644 tests/megatron/test_accuracy_loss_and_norm.py create mode 100644 tests/megatron/test_model_repro_machine_outputs.py diff --git a/swift/megatron/callbacks/print.py b/swift/megatron/callbacks/print.py index 4b8a897f68..8c5e371ebd 100644 --- a/swift/megatron/callbacks/print.py +++ b/swift/megatron/callbacks/print.py @@ -1,7 +1,12 @@ # Copyright (c) ModelScope Contributors. All rights reserved. +import hashlib +import json import os +import platform +import sys import time import torch +from pathlib import Path from tqdm import tqdm from swift.megatron.utils import reduce_max_stat_across_model_parallel_group @@ -20,6 +25,136 @@ def raw_loss_event(step, logs): return {'step': step, **raw_losses} if raw_losses else None +def _sha256_file(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def normalized_device(): + """Return the device class the benchmark checker expects, not the GPU model name.""" + return 'cuda' if torch.cuda.is_available() else 'cpu' + + +def normalized_dtype(value): + """Return the bench dtype alias; a bare ``torch.bfloat16`` string is rejected.""" + text = str(value or '').strip() + prefix = 'torch.' + if text.startswith(prefix): + text = text[len(prefix):] + return text.lower() + + +def machine_loss_payload(events, raw_path=None, source_sha256=None): + """Return the machine loss artifact. + + ``losses`` is the benchmark gate field: an unrounded main-loss series with one + entry per recorded step. ``events`` keeps the per-step diagnostic detail. + """ + return { + 'schema': 'glm52-machine-loss/v1', + 'framework': 'torch', + 'raw': True, + 'stage': 'training_callback_complete', + 'losses': [event['loss'] for event in events if 'loss' in event], + 'event_count': len(events), + 'steps': [event['step'] for event in events], + 'events': events, + 'source': raw_path, + 'source_sha256': source_sha256, + } + + +def _write_json(path, payload): + path = Path(path).expanduser().resolve() + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(payload, ensure_ascii=False, allow_nan=False, indent=2, sort_keys=True) + '\n') + + +def model_repro_source_modules(): + names = ( + 'swift.megatron.callbacks.print', + 'swift.megatron.trainers.base', + 'megatron.core.transformer.moe.moe_utils', + 'megatron.core.transformer.moe.experts', + 'megatron.core.transformer.moe.router', + 'megatron.core.models.common.language_module.language_module', + 'mcore_bridge.model.gpt_model', + ) + records = {} + for name in names: + path = getattr(sys.modules.get(name), '__file__', None) + records[name] = { + 'path': str(Path(path).resolve()) if path else None, + 'sha256': _sha256_file(path) if path and Path(path).is_file() else None + } + return records + + +def model_repro_topology(args, config): + from megatron.core import parallel_state + + fields = ('tensor_model_parallel_size', 'pipeline_model_parallel_size', 'expert_model_parallel_size', + 'expert_tensor_parallel_size', 'context_parallel_size', 'sequence_parallel') + record = { + 'configured': { + name: getattr(config, name, None) + for name in fields + }, + 'world_size': torch.distributed.get_world_size(), + 'rank': torch.distributed.get_rank(), + 'global_batch_size': args.global_batch_size, + 'micro_batch_size': args.micro_batch_size, + 'groups': {} + } + getters = { + 'tp': 'get_tensor_model_parallel_group', + 'pp': 'get_pipeline_model_parallel_group', + 'ep': 'get_expert_model_parallel_group', + 'etp': 'get_expert_tensor_parallel_group', + 'cp': 'get_context_parallel_group', + 'data': 'get_data_parallel_group' + } + for name, getter in getters.items(): + group = getattr(parallel_state, getter)() + record['groups'][name] = { + 'size': torch.distributed.get_world_size(group), + 'ranks': torch.distributed.get_process_group_ranks(group) + } + return record + + +def model_repro_environment(args, trainer): + """Return the formal run-local environment receipt after model loading.""" + config_path = os.environ.get('MODEL_REPRO_MODEL_CONFIG_PATH') + return { + 'schema': 'glm52-environment/v1', + 'framework': 'torch', + 'framework_version': torch.__version__, + 'python_version': platform.python_version(), + 'device': normalized_device(), + 'device_name': torch.cuda.get_device_name(torch.cuda.current_device()), + 'dtype': normalized_dtype(getattr(args, 'torch_dtype', 'bfloat16')), + 'cuda': torch.version.cuda, + 'cudnn': torch.backends.cudnn.version(), + 'nccl': list(torch.cuda.nccl.version()), + 'deterministic': { + 'algorithms_enabled': torch.are_deterministic_algorithms_enabled(), + 'cudnn_deterministic': torch.backends.cudnn.deterministic, + 'cudnn_benchmark': torch.backends.cudnn.benchmark, + 'cublas_workspace_config': os.environ.get('CUBLAS_WORKSPACE_CONFIG'), + 'nccl_algo': os.environ.get('NCCL_ALGO'), + }, + 'model_id': os.environ.get('MODEL_REPRO_MODEL_ID'), + 'revision': os.environ.get('MODEL_REPRO_MODEL_REVISION'), + 'model_config_sha256': _sha256_file(config_path) if config_path else None, + 'weights_loaded': bool(getattr(trainer, '_model_repro_weights_source', None)), + 'model_source': str(getattr(trainer, '_model_repro_weights_source', '')), + 'topology': model_repro_topology(args, trainer.config), + 'source_modules': model_repro_source_modules(), + 'invocation_id': os.environ.get('MRK_INVOCATION_ID'), + 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, + } + + class PrintCallback(MegatronCallback): def __init__(self, trainer): @@ -28,6 +163,7 @@ def __init__(self, trainer): self.eval_bar = None self.jsonl_writer = None self.raw_loss_writer = None + self.raw_loss_events = [] self.is_write_rank = is_last_rank() def on_train_begin(self): @@ -37,17 +173,35 @@ def on_train_begin(self): self.training_bar.update(self.state.iteration) self.current_step = self.state.iteration self.start_time = time.time() + self.raw_loss_events = [] + loss_path = os.environ.get('MODEL_REPRO_LOSS_PATH') + if loss_path and self.is_write_rank: + Path(loss_path).expanduser().resolve().unlink(missing_ok=True) logging_path = os.path.join(self.args.output_dir, 'logging.jsonl') logger.info(f'logging_path: {logging_path}') self.jsonl_writer = JsonlWriter(logging_path, enable_async=True, write_on_rank='last') raw_loss_path = os.environ.get('MODEL_REPRO_RAW_LOSS_PATH') if raw_loss_path: logger.info(f'raw_loss_path: {raw_loss_path}') + if self.is_write_rank and Path(raw_loss_path).exists(): + Path(raw_loss_path).unlink() self.raw_loss_writer = JsonlWriter(raw_loss_path, write_on_rank='last') + env_path = os.environ.get('MODEL_REPRO_ENV_PATH') + if env_path and self.is_write_rank: + _write_json(env_path, model_repro_environment(self.args, self.trainer)) def on_train_end(self): self.training_bar.close() self.training_bar = None + loss_path = os.environ.get('MODEL_REPRO_LOSS_PATH') + if loss_path and self.is_write_rank: + raw_path = os.environ.get('MODEL_REPRO_RAW_LOSS_PATH') + payload = machine_loss_payload( + self.raw_loss_events, + raw_path=raw_path, + source_sha256=_sha256_file(raw_path) if raw_path and Path(raw_path).is_file() else None, + ) + _write_json(loss_path, payload) def on_step_end(self): n_step = self.state.iteration - self.current_step @@ -80,6 +234,8 @@ def on_log(self, logs): raw_event = raw_loss_event(state.iteration, logs) if self.raw_loss_writer is not None and raw_event is not None: self.raw_loss_writer.append(raw_event) + if raw_event is not None and self.is_write_rank and os.environ.get('MODEL_REPRO_LOSS_PATH'): + self.raw_loss_events.append(raw_event) logs = {k: round(v, 8) if isinstance(v, float) else v for k, v in logs.items()} self.jsonl_writer.append(logs) if self.is_write_rank: diff --git a/swift/megatron/init.py b/swift/megatron/init.py index 721b811590..df07c34b7b 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -277,6 +277,29 @@ def _set_layer_mlp(self, mg_layer, hf_state_dict, layer_idx, to_mcore, is_mtp=Fa ) +def _patch_mcore_bridge_tp1_accuracy(): + """Keep the TP1 accuracy graph free of bridge-only viewless nodes.""" + from megatron.core import parallel_state + from mcore_bridge.model.modules import mtp_layer, transformer_block + + def patch_module(module): + original = module.make_viewless_tensor + if getattr(original, '_swift_tp1_accuracy_patch', False): + return + + def make_viewless_tensor(inp, requires_grad, keep_graph): + if (_use_accuracy_compatible_enabled() + and parallel_state.get_tensor_model_parallel_world_size() <= 1): + return inp + return original(inp=inp, requires_grad=requires_grad, keep_graph=keep_graph) + + make_viewless_tensor._swift_tp1_accuracy_patch = True + module.make_viewless_tensor = make_viewless_tensor + + patch_module(mtp_layer) + patch_module(transformer_block) + + def _patch_mcore_bridge(): require_version( 'mcore-bridge>=1.4.0', @@ -288,18 +311,26 @@ def _patch_mcore_bridge(): logger.info(f'mcore_bridge.__version__: {mcore_bridge.__version__}') if _use_accuracy_compatible_enabled(): _patch_mcore_bridge_disable_te() + _patch_mcore_bridge_tp1_accuracy() if not getattr(ModelLoader._replace_spec_dsa, '_swift_norm_accuracy_patch', False): origin_replace_spec_dsa = ModelLoader._replace_spec_dsa def replace_spec_dsa(self, layer_spec): origin_replace_spec_dsa(self, layer_spec) - if not getattr(self.config, 'norm_accuracy_compatible', False): - return from megatron.core.transformer.torch_norm import WrappedTorchNorm dsa_spec = layer_spec.submodules.self_attention - dsa_spec.submodules.q_layernorm = WrappedTorchNorm - dsa_spec.submodules.kv_layernorm = WrappedTorchNorm + if getattr(self.config, 'norm_accuracy_compatible', False): + dsa_spec.submodules.q_layernorm = WrappedTorchNorm + dsa_spec.submodules.kv_layernorm = WrappedTorchNorm + indexer = getattr( + getattr(dsa_spec.submodules.core_attention, 'submodules', None), + 'indexer', + None, + ) + if (_use_accuracy_compatible_enabled() and indexer is not None + and getattr(indexer, 'submodules', None) is not None): + indexer.submodules.k_norm = WrappedTorchNorm replace_spec_dsa._swift_norm_accuracy_patch = True ModelLoader._replace_spec_dsa = replace_spec_dsa diff --git a/swift/megatron/trainers/base.py b/swift/megatron/trainers/base.py index a4e4349f5b..0b0d5da9ad 100644 --- a/swift/megatron/trainers/base.py +++ b/swift/megatron/trainers/base.py @@ -116,6 +116,7 @@ def _load_checkpoint(self): if args.mcore_model is not None: self.state.iteration = load_mcore_checkpoint( args, self.wrapped_models, self.optimizer, self.opt_param_scheduler, load_arg='mcore_model') + self._model_repro_weights_source = args.mcore_model if args.mcore_adapter is not None: self.state.iteration = load_mcore_checkpoint( args, self.wrapped_models, self.optimizer, self.opt_param_scheduler, load_arg='mcore_adapter') @@ -199,6 +200,7 @@ def _prepare_peft_model(self, models): args = self.args if args.mcore_model is None: self.bridge.load_weights(models, args.model_dir) + self._model_repro_weights_source = args.model_dir peft_models = [prepare_mcore_model(args, model) for model in models] if args.tuner_type == 'lora' and args.adapters and args.mcore_adapter is None: assert len(args.adapters) == 1, 'Currently only support one adapter.' @@ -776,9 +778,13 @@ def save_checkpoint(self): state = self.state args.consumed_train_samples = state.consumed_train_samples iteration = state.iteration - output_dir = os.path.join(args.output_dir, f'checkpoint-{iteration}') + formal_checkpoint_dir = os.environ.get('MODEL_REPRO_CHECKPOINT_DIR') + if formal_checkpoint_dir and iteration == args.train_iters: + output_dir = os.path.abspath(os.path.expanduser(formal_checkpoint_dir)) + else: + output_dir = os.path.join(args.output_dir, f'checkpoint-{iteration}') os.makedirs(output_dir, exist_ok=True) - args_path = os.path.join(os.path.dirname(output_dir), 'args.json') + args_path = os.path.join(args.output_dir, 'args.json') self.copy_path(args_path, os.path.join(output_dir, 'args.json')) if args.save_safetensors and args.no_save_optim: model = [] diff --git a/swift/megatron/trainers/trainer.py b/swift/megatron/trainers/trainer.py index a7da085ef2..62a77c69d5 100644 --- a/swift/megatron/trainers/trainer.py +++ b/swift/megatron/trainers/trainer.py @@ -385,12 +385,7 @@ def loss_func(self, losses = losses * torch.exp(-losses.detach()) if loss_scale is not None: losses = losses * loss_scale - from megatron.core.transformer.module import _use_accuracy_compatible - if _use_accuracy_compatible(): - loss_sum = (losses * loss_mask).reshape(-1).double().sum().float() - else: - loss_sum = torch.sum(losses * loss_mask) - loss = torch.cat([loss_sum.view(1), loss_mask.sum().view(1)]) + loss = torch.cat([torch.sum(losses * loss_mask).view(1), loss_mask.sum().view(1)]) # Reduce loss for logging. reporting_loss = loss.detach().clone() diff --git a/tests/megatron/test_accuracy_bridge_tp1.py b/tests/megatron/test_accuracy_bridge_tp1.py new file mode 100644 index 0000000000..c3f532082d --- /dev/null +++ b/tests/megatron/test_accuracy_bridge_tp1.py @@ -0,0 +1,123 @@ +"""Unit tests for ms-swift TP1 accuracy bridge patch.""" +from __future__ import annotations + +import ast +import sys +import unittest +from pathlib import Path +from types import ModuleType, SimpleNamespace +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +_UAC = {"on": False} +_TP = {"size": 1} + + +def _use_accuracy_compatible_enabled(): + return _UAC["on"] + + +def _load_patch(): + src = (ROOT / "swift/megatron/init.py").read_text() + tree = ast.parse(src) + target = next( + node + for node in tree.body + if isinstance(node, ast.FunctionDef) and node.name == "_patch_mcore_bridge_tp1_accuracy" + ) + target.decorator_list = [] + mod = ast.Module(body=[target], type_ignores=[]) + ast.fix_missing_locations(mod) + ns = {"_use_accuracy_compatible_enabled": _use_accuracy_compatible_enabled} + exec(compile(mod, "swift/megatron/init.py", "exec"), ns) + return ns["_patch_mcore_bridge_tp1_accuracy"] + + +_patch_mcore_bridge_tp1_accuracy = _load_patch() + + +def _make_original(name): + def make_viewless_tensor(inp, requires_grad, keep_graph): + make_viewless_tensor.calls.append((inp, requires_grad, keep_graph)) + return ("wrapped", inp) + + make_viewless_tensor.calls = [] + make_viewless_tensor.__name__ = name + return make_viewless_tensor + + +class TestBridgeTp1Patch(unittest.TestCase): + def setUp(self): + self.orig_mtp = _make_original("mtp") + self.orig_block = _make_original("block") + self.mtp = ModuleType("mtp_layer") + self.block = ModuleType("transformer_block") + self.mtp.make_viewless_tensor = self.orig_mtp + self.block.make_viewless_tensor = self.orig_block + + fake_ps = SimpleNamespace( + get_tensor_model_parallel_world_size=lambda: _TP["size"] + ) + fake_core = ModuleType("megatron.core") + fake_core.parallel_state = fake_ps + fake_mcore = ModuleType("mcore_bridge") + fake_model = ModuleType("mcore_bridge.model") + fake_modules = ModuleType("mcore_bridge.model.modules") + fake_modules.mtp_layer = self.mtp + fake_modules.transformer_block = self.block + fake_megatron = ModuleType("megatron") + + p = patch.dict( + sys.modules, + { + "megatron": fake_megatron, + "megatron.core": fake_core, + "mcore_bridge": fake_mcore, + "mcore_bridge.model": fake_model, + "mcore_bridge.model.modules": fake_modules, + }, + ) + p.start() + self.addCleanup(p.stop) + + def tearDown(self): + _UAC["on"] = False + _TP["size"] = 1 + + def test_idempotent_and_both_module_globals(self): + _patch_mcore_bridge_tp1_accuracy() + first_mtp = self.mtp.make_viewless_tensor + first_block = self.block.make_viewless_tensor + self.assertTrue(getattr(first_mtp, "_swift_tp1_accuracy_patch", False)) + self.assertTrue(getattr(first_block, "_swift_tp1_accuracy_patch", False)) + _patch_mcore_bridge_tp1_accuracy() + self.assertIs(self.mtp.make_viewless_tensor, first_mtp) + self.assertIs(self.block.make_viewless_tensor, first_block) + + def test_uac_tp1_identity_both_modules(self): + _patch_mcore_bridge_tp1_accuracy() + _UAC["on"] = True + _TP["size"] = 1 + inp = object() + self.assertIs(self.mtp.make_viewless_tensor(inp, True, True), inp) + self.assertIs(self.block.make_viewless_tensor(inp, False, False), inp) + self.assertEqual(self.orig_mtp.calls, []) + self.assertEqual(self.orig_block.calls, []) + + def test_off_and_actual_tp2_delegate_to_original(self): + _patch_mcore_bridge_tp1_accuracy() + inp = object() + _UAC["on"] = False + _TP["size"] = 1 + out = self.mtp.make_viewless_tensor(inp, True, True) + self.assertEqual(out, ("wrapped", inp)) + self.assertEqual(self.orig_mtp.calls[-1], (inp, True, True)) + _UAC["on"] = True + _TP["size"] = 2 + out2 = self.block.make_viewless_tensor(inp, False, True) + self.assertEqual(out2, ("wrapped", inp)) + self.assertEqual(self.orig_block.calls[-1], (inp, False, True)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/megatron/test_accuracy_loss_and_norm.py b/tests/megatron/test_accuracy_loss_and_norm.py new file mode 100644 index 0000000000..46c1cec583 --- /dev/null +++ b/tests/megatron/test_accuracy_loss_and_norm.py @@ -0,0 +1,89 @@ +"""Native criterion and DSA specification contracts without distributed setup.""" + +import ast +import contextlib +import io +import sys +import types +import unittest +from pathlib import Path +from unittest.mock import patch + +import torch + +ROOT = Path(__file__).resolve().parents[2] + + +def production_function(relative_path, name, namespace): + path = ROOT / relative_path + tree = ast.parse(path.read_text()) + matches = [node for node in ast.walk(tree) if isinstance(node, ast.FunctionDef) and node.name == name] + if len(matches) != 1: + raise AssertionError(f'Expected one production function: {path}:{name}') + node = matches[0] + node.decorator_list = [] + module = ast.Module(body=[ast.ImportFrom(module='__future__', names=[ast.alias(name='annotations')], level=0), node], + type_ignores=[]) + exec(compile(ast.fix_missing_locations(module), str(path), 'exec'), namespace) + return namespace[name] + + +class AccuracyLossAndNormTest(unittest.TestCase): + + def test_mask_scale_local_gradient_and_global_reporting(self): + torch.cuda.set_device(0) + for enabled in (False, True): + with self.subTest(accuracy=enabled): + loss_func = production_function( + 'swift/megatron/trainers/trainer.py', 'loss_func', { + 'torch': torch, + 'mpu': types.SimpleNamespace(get_data_parallel_group=lambda **kwargs: None), + '_use_accuracy_compatible_enabled': lambda: enabled, + }) + trainer = types.SimpleNamespace(args=types.SimpleNamespace(enable_dft_loss=False, + enable_channel_loss=False)) + values = torch.tensor([[2., 19., 3.]], device='cuda', requires_grad=True) + labels = torch.tensor([[1, -100, 2]], device='cuda') + scale = torch.tensor([[0.5, 1000., 2.]], device='cuda') + # A second identical DP rank contributes to reporting, not local backward. + with patch.object(torch.distributed, 'all_reduce', side_effect=lambda value, **kwargs: value.mul_(2)), \ + patch.object(torch.distributed, 'get_rank', return_value=0), \ + contextlib.redirect_stdout(io.StringIO()): + loss, count, metrics = loss_func(trainer, values, labels=labels, loss_scale=scale) + self.assertEqual(loss.item(), 7.) + self.assertEqual(count.item(), 2) + self.assertEqual(metrics['loss'].tolist(), [14., 4.]) + self.assertFalse(metrics['loss'].requires_grad) + loss.backward() + self.assertEqual(values.grad.tolist(), [[0.5, 0., 2.]]) + + def test_indexer_norm_preserves_disabled_provider(self): + native_norm = type('NativeNorm', (), {}) + provider_norm = type('ProviderNorm', (), {}) + module = types.ModuleType('megatron.core.transformer.torch_norm') + module.WrappedTorchNorm = native_norm + for enabled, norm_accuracy in ((False, False), (True, False), (True, True)): + with self.subTest(accuracy=enabled, norm_accuracy=norm_accuracy): + calls = [] + replace_spec = production_function( + 'swift/megatron/init.py', 'replace_spec_dsa', { + 'origin_replace_spec_dsa': lambda *args: calls.append('provider'), + '_use_accuracy_compatible_enabled': lambda: enabled, + }) + indexer = types.SimpleNamespace(submodules=types.SimpleNamespace(k_norm=provider_norm)) + attention = types.SimpleNamespace(submodules=types.SimpleNamespace( + q_layernorm=provider_norm, kv_layernorm=provider_norm, + core_attention=types.SimpleNamespace(submodules=types.SimpleNamespace(indexer=indexer)))) + spec = types.SimpleNamespace(submodules=types.SimpleNamespace(self_attention=attention)) + loader = types.SimpleNamespace(config=types.SimpleNamespace(norm_accuracy_compatible=norm_accuracy)) + with patch.dict(sys.modules, {module.__name__: module}): + replace_spec(loader, spec) + self.assertEqual(calls, ['provider']) + self.assertIs(indexer.submodules.k_norm, native_norm if enabled else provider_norm) + expected_qkv = native_norm if norm_accuracy else provider_norm + self.assertIs(attention.submodules.q_layernorm, expected_qkv) + self.assertIs(attention.submodules.kv_layernorm, expected_qkv) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/megatron/test_model_repro_machine_outputs.py b/tests/megatron/test_model_repro_machine_outputs.py new file mode 100644 index 0000000000..75fccb4410 --- /dev/null +++ b/tests/megatron/test_model_repro_machine_outputs.py @@ -0,0 +1,74 @@ +# Copyright (c) ModelScope Contributors. All rights reserved. +"""Production machine-output and checkpoint-path contracts without GPU imports.""" +import ast +import hashlib +import json +import os +import tempfile +import unittest +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import patch + +ROOT = Path(__file__).resolve().parents[2] +PRINT = ROOT / 'swift/megatron/callbacks/print.py' +BASE = ROOT / 'swift/megatron/trainers/base.py' +namespace = {'os': os, 'Path': Path, 'hashlib': hashlib, 'json': json} +functions = [ + n for n in ast.parse(PRINT.read_text()).body + if isinstance(n, ast.FunctionDef) and n.name in ('raw_loss_event', 'machine_loss_payload', '_write_json') +] +exec(compile(ast.Module(body=functions, type_ignores=[]), str(PRINT), 'exec'), namespace) +trainer = next( + n for n in ast.parse(BASE.read_text()).body if isinstance(n, ast.ClassDef) and n.name == 'BaseMegatronTrainer') +save = next(n for n in trainer.body if isinstance(n, ast.FunctionDef) and n.name == 'save_checkpoint') +namespace.update(gc_collect=lambda: None, save_mcore_checkpoint=lambda *args, **kwargs: None, is_master=lambda: False) +exec(compile(ast.Module(body=[save], type_ignores=[]), str(BASE), 'exec'), namespace) + + +class MachineOutputTests(unittest.TestCase): + + def test_unrounded_training_losses_exclude_evaluation(self): + event = namespace['raw_loss_event'](1, {'loss': 0.123456789123, 'mtp_1_loss': 0.5, 'eval_loss': 10}) + self.assertEqual(event, {'step': 1, 'loss': 0.123456789123, 'mtp_1_loss': 0.5}) + self.assertIsNone(namespace['raw_loss_event'](2, {'eval_loss': 10})) + result = namespace['machine_loss_payload']([event]) + self.assertEqual(result['losses'], [0.123456789123]) + self.assertEqual(result['steps'], [1]) + self.assertNotIn('owning_cli_exit_code', result) + + def test_final_override_preserves_native_export_and_args_source(self): + for iteration, override in [(99, True), (100, True), (100, False)]: + with self.subTest(iteration=iteration, override=override), tempfile.TemporaryDirectory() as directory: + output = str(Path(directory) / 'training') + final = str(Path(directory) / 'canonical' / 'checkpoint') + args = SimpleNamespace( + output_dir=output, + train_iters=100, + save_safetensors=True, + no_save_optim=True, + tuner_type='full', + merge_lora=False) + state = SimpleNamespace(iteration=iteration, consumed_train_samples=iteration, best_global_step=None) + copies, exports = [], [] + obj = SimpleNamespace( + args=args, + state=state, + optimizer=None, + opt_param_scheduler=None, + unwrapped_models=['native-model'], + template=SimpleNamespace(processor='processor'), + bridge=SimpleNamespace(save_weights=lambda *a, **kw: exports.append((a, kw))), + copy_path=lambda source, target: copies.append((source, target))) + environment = {'MODEL_REPRO_CHECKPOINT_DIR': final} if override else {} + with patch.dict(os.environ, environment, clear=True): + namespace['save_checkpoint'](obj) + expected = final if override and iteration == 100 else str(Path(output) / f'checkpoint-{iteration}') + self.assertEqual(state.last_model_checkpoint, expected) + self.assertEqual(copies[0], (str(Path(output) / 'args.json'), str(Path(expected) / 'args.json'))) + self.assertEqual(exports[0][0], (['native-model'], expected)) + self.assertEqual(exports[0][1]['processor'], 'processor') + + +if __name__ == '__main__': + unittest.main() From 06c3eca0a2f3b44179ed9cab74a3255c0e232a02 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Wed, 9 Sep 2026 03:31:00 +0800 Subject: [PATCH 25/33] style: satisfy repository format checks for GLM52 alignment --- swift/megatron/init.py | 5 +- tests/megatron/test_accuracy_bridge_tp1.py | 82 +++++++++---------- tests/megatron/test_accuracy_loss_and_norm.py | 20 +++-- 3 files changed, 53 insertions(+), 54 deletions(-) diff --git a/swift/megatron/init.py b/swift/megatron/init.py index df07c34b7b..7953834558 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -279,8 +279,8 @@ def _set_layer_mlp(self, mg_layer, hf_state_dict, layer_idx, to_mcore, is_mtp=Fa def _patch_mcore_bridge_tp1_accuracy(): """Keep the TP1 accuracy graph free of bridge-only viewless nodes.""" - from megatron.core import parallel_state from mcore_bridge.model.modules import mtp_layer, transformer_block + from megatron.core import parallel_state def patch_module(module): original = module.make_viewless_tensor @@ -288,8 +288,7 @@ def patch_module(module): return def make_viewless_tensor(inp, requires_grad, keep_graph): - if (_use_accuracy_compatible_enabled() - and parallel_state.get_tensor_model_parallel_world_size() <= 1): + if (_use_accuracy_compatible_enabled() and parallel_state.get_tensor_model_parallel_world_size() <= 1): return inp return original(inp=inp, requires_grad=requires_grad, keep_graph=keep_graph) diff --git a/tests/megatron/test_accuracy_bridge_tp1.py b/tests/megatron/test_accuracy_bridge_tp1.py index c3f532082d..30a261b219 100644 --- a/tests/megatron/test_accuracy_bridge_tp1.py +++ b/tests/megatron/test_accuracy_bridge_tp1.py @@ -9,37 +9,36 @@ from unittest.mock import patch ROOT = Path(__file__).resolve().parents[2] -_UAC = {"on": False} -_TP = {"size": 1} +_UAC = {'on': False} +_TP = {'size': 1} def _use_accuracy_compatible_enabled(): - return _UAC["on"] + return _UAC['on'] def _load_patch(): - src = (ROOT / "swift/megatron/init.py").read_text() + src = (ROOT / 'swift/megatron/init.py').read_text() tree = ast.parse(src) target = next( - node - for node in tree.body - if isinstance(node, ast.FunctionDef) and node.name == "_patch_mcore_bridge_tp1_accuracy" - ) + node for node in tree.body + if isinstance(node, ast.FunctionDef) and node.name == '_patch_mcore_bridge_tp1_accuracy') target.decorator_list = [] mod = ast.Module(body=[target], type_ignores=[]) ast.fix_missing_locations(mod) - ns = {"_use_accuracy_compatible_enabled": _use_accuracy_compatible_enabled} - exec(compile(mod, "swift/megatron/init.py", "exec"), ns) - return ns["_patch_mcore_bridge_tp1_accuracy"] + ns = {'_use_accuracy_compatible_enabled': _use_accuracy_compatible_enabled} + exec(compile(mod, 'swift/megatron/init.py', 'exec'), ns) + return ns['_patch_mcore_bridge_tp1_accuracy'] _patch_mcore_bridge_tp1_accuracy = _load_patch() def _make_original(name): + def make_viewless_tensor(inp, requires_grad, keep_graph): make_viewless_tensor.calls.append((inp, requires_grad, keep_graph)) - return ("wrapped", inp) + return ('wrapped', inp) make_viewless_tensor.calls = [] make_viewless_tensor.__name__ = name @@ -47,57 +46,56 @@ def make_viewless_tensor(inp, requires_grad, keep_graph): class TestBridgeTp1Patch(unittest.TestCase): + def setUp(self): - self.orig_mtp = _make_original("mtp") - self.orig_block = _make_original("block") - self.mtp = ModuleType("mtp_layer") - self.block = ModuleType("transformer_block") + self.orig_mtp = _make_original('mtp') + self.orig_block = _make_original('block') + self.mtp = ModuleType('mtp_layer') + self.block = ModuleType('transformer_block') self.mtp.make_viewless_tensor = self.orig_mtp self.block.make_viewless_tensor = self.orig_block - fake_ps = SimpleNamespace( - get_tensor_model_parallel_world_size=lambda: _TP["size"] - ) - fake_core = ModuleType("megatron.core") + fake_ps = SimpleNamespace(get_tensor_model_parallel_world_size=lambda: _TP['size']) + fake_core = ModuleType('megatron.core') fake_core.parallel_state = fake_ps - fake_mcore = ModuleType("mcore_bridge") - fake_model = ModuleType("mcore_bridge.model") - fake_modules = ModuleType("mcore_bridge.model.modules") + fake_mcore = ModuleType('mcore_bridge') + fake_model = ModuleType('mcore_bridge.model') + fake_modules = ModuleType('mcore_bridge.model.modules') fake_modules.mtp_layer = self.mtp fake_modules.transformer_block = self.block - fake_megatron = ModuleType("megatron") + fake_megatron = ModuleType('megatron') p = patch.dict( sys.modules, { - "megatron": fake_megatron, - "megatron.core": fake_core, - "mcore_bridge": fake_mcore, - "mcore_bridge.model": fake_model, - "mcore_bridge.model.modules": fake_modules, + 'megatron': fake_megatron, + 'megatron.core': fake_core, + 'mcore_bridge': fake_mcore, + 'mcore_bridge.model': fake_model, + 'mcore_bridge.model.modules': fake_modules, }, ) p.start() self.addCleanup(p.stop) def tearDown(self): - _UAC["on"] = False - _TP["size"] = 1 + _UAC['on'] = False + _TP['size'] = 1 def test_idempotent_and_both_module_globals(self): _patch_mcore_bridge_tp1_accuracy() first_mtp = self.mtp.make_viewless_tensor first_block = self.block.make_viewless_tensor - self.assertTrue(getattr(first_mtp, "_swift_tp1_accuracy_patch", False)) - self.assertTrue(getattr(first_block, "_swift_tp1_accuracy_patch", False)) + self.assertTrue(getattr(first_mtp, '_swift_tp1_accuracy_patch', False)) + self.assertTrue(getattr(first_block, '_swift_tp1_accuracy_patch', False)) _patch_mcore_bridge_tp1_accuracy() self.assertIs(self.mtp.make_viewless_tensor, first_mtp) self.assertIs(self.block.make_viewless_tensor, first_block) def test_uac_tp1_identity_both_modules(self): _patch_mcore_bridge_tp1_accuracy() - _UAC["on"] = True - _TP["size"] = 1 + _UAC['on'] = True + _TP['size'] = 1 inp = object() self.assertIs(self.mtp.make_viewless_tensor(inp, True, True), inp) self.assertIs(self.block.make_viewless_tensor(inp, False, False), inp) @@ -107,17 +105,17 @@ def test_uac_tp1_identity_both_modules(self): def test_off_and_actual_tp2_delegate_to_original(self): _patch_mcore_bridge_tp1_accuracy() inp = object() - _UAC["on"] = False - _TP["size"] = 1 + _UAC['on'] = False + _TP['size'] = 1 out = self.mtp.make_viewless_tensor(inp, True, True) - self.assertEqual(out, ("wrapped", inp)) + self.assertEqual(out, ('wrapped', inp)) self.assertEqual(self.orig_mtp.calls[-1], (inp, True, True)) - _UAC["on"] = True - _TP["size"] = 2 + _UAC['on'] = True + _TP['size'] = 2 out2 = self.block.make_viewless_tensor(inp, False, True) - self.assertEqual(out2, ("wrapped", inp)) + self.assertEqual(out2, ('wrapped', inp)) self.assertEqual(self.orig_block.calls[-1], (inp, False, True)) -if __name__ == "__main__": +if __name__ == '__main__': unittest.main() diff --git a/tests/megatron/test_accuracy_loss_and_norm.py b/tests/megatron/test_accuracy_loss_and_norm.py index 46c1cec583..506c00de52 100644 --- a/tests/megatron/test_accuracy_loss_and_norm.py +++ b/tests/megatron/test_accuracy_loss_and_norm.py @@ -4,13 +4,12 @@ import contextlib import io import sys +import torch import types import unittest from pathlib import Path from unittest.mock import patch -import torch - ROOT = Path(__file__).resolve().parents[2] @@ -22,8 +21,9 @@ def production_function(relative_path, name, namespace): raise AssertionError(f'Expected one production function: {path}:{name}') node = matches[0] node.decorator_list = [] - module = ast.Module(body=[ast.ImportFrom(module='__future__', names=[ast.alias(name='annotations')], level=0), node], - type_ignores=[]) + module = ast.Module( + body=[ast.ImportFrom(module='__future__', names=[ast.alias(name='annotations')], level=0), node], + type_ignores=[]) exec(compile(ast.fix_missing_locations(module), str(path), 'exec'), namespace) return namespace[name] @@ -40,8 +40,8 @@ def test_mask_scale_local_gradient_and_global_reporting(self): 'mpu': types.SimpleNamespace(get_data_parallel_group=lambda **kwargs: None), '_use_accuracy_compatible_enabled': lambda: enabled, }) - trainer = types.SimpleNamespace(args=types.SimpleNamespace(enable_dft_loss=False, - enable_channel_loss=False)) + trainer = types.SimpleNamespace( + args=types.SimpleNamespace(enable_dft_loss=False, enable_channel_loss=False)) values = torch.tensor([[2., 19., 3.]], device='cuda', requires_grad=True) labels = torch.tensor([[1, -100, 2]], device='cuda') scale = torch.tensor([[0.5, 1000., 2.]], device='cuda') @@ -71,9 +71,11 @@ def test_indexer_norm_preserves_disabled_provider(self): '_use_accuracy_compatible_enabled': lambda: enabled, }) indexer = types.SimpleNamespace(submodules=types.SimpleNamespace(k_norm=provider_norm)) - attention = types.SimpleNamespace(submodules=types.SimpleNamespace( - q_layernorm=provider_norm, kv_layernorm=provider_norm, - core_attention=types.SimpleNamespace(submodules=types.SimpleNamespace(indexer=indexer)))) + attention = types.SimpleNamespace( + submodules=types.SimpleNamespace( + q_layernorm=provider_norm, + kv_layernorm=provider_norm, + core_attention=types.SimpleNamespace(submodules=types.SimpleNamespace(indexer=indexer)))) spec = types.SimpleNamespace(submodules=types.SimpleNamespace(self_attention=attention)) loader = types.SimpleNamespace(config=types.SimpleNamespace(norm_accuracy_compatible=norm_accuracy)) with patch.dict(sys.modules, {module.__name__: module}): From 57464e021c867ceecd9d3689e70892c8a515267c Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Wed, 9 Sep 2026 13:04:43 +0800 Subject: [PATCH 26/33] Remove unrelated CI changes from GLM alignment PR --- .../workflows/alignment_model_accuracy.yaml | 164 +--- .github/workflows/citest.yaml | 16 +- scripts/consume_paddlefleet_alignment_pin.sh | 370 -------- scripts/dependence/build.sh | 1 + scripts/require_paddlefleet_selector_ok.sh | 40 - scripts/select_paddlefleet_alignment_pin.sh | 889 ------------------ scripts/test_alignment_workflow_shell.sh | 161 ---- scripts/test_checkout_no_pager.sh | 65 -- scripts/test_checkout_observability.sh | 60 -- scripts/test_paddlefleet_pin_handoff.sh | 114 --- 10 files changed, 33 insertions(+), 1847 deletions(-) delete mode 100755 scripts/consume_paddlefleet_alignment_pin.sh delete mode 100755 scripts/require_paddlefleet_selector_ok.sh delete mode 100755 scripts/select_paddlefleet_alignment_pin.sh delete mode 100755 scripts/test_alignment_workflow_shell.sh delete mode 100755 scripts/test_checkout_no_pager.sh delete mode 100755 scripts/test_checkout_observability.sh delete mode 100755 scripts/test_paddlefleet_pin_handoff.sh diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index cbf4d60d6b..76faf444ba 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -3,36 +3,6 @@ name: Alignment Model Accuracy on: pull_request: workflow_dispatch: - inputs: - paddlefleet_mode: - description: "develop keeps the historical CodeSync tarball. stack-paired fail-closed SHA check." - type: choice - default: develop - options: [develop, stack-paired] - paddlefleet_pin_sha: - description: "40-hex PaddleFleet commit (required for stack-paired)" - required: false - type: string - paddlefleet_git_url: - description: "Git URL that contains paddlefleet_pin_sha" - required: false - type: string - paddlefleet_wheel_url: - description: "Optional wheel URL from Build Fleet whl / Actions artifact metadata" - required: false - type: string - paddlefleet_wheel_sha256: - description: "sha256 for paddlefleet_wheel_url" - required: false - type: string - paddlefleet_ops_wheel_url: - description: "Optional paddlefleet_ops wheel URL" - required: false - type: string - paddlefleet_ops_wheel_sha256: - description: "sha256 for paddlefleet_ops_wheel_url" - required: false - type: string concurrency: group: Alignment-${{ github.workflow }}-${{ github.event.pull_request.number }} @@ -47,14 +17,6 @@ env: TASK: MS-SWIFT-${{ github.sha }}-alignment CE_name: alignment-ms-swift no_proxy: "localhost,bj.bcebos.com,su.bcebos.com,bcebos.com,apiin.im.baidu.com,gitee.com,aliyun.com,.baidu.com,.tuna.tsinghua.edu.cn" - # pull_request stays on historical develop. stack-paired is workflow_dispatch only. - ALIGNMENT_PADDLEFLEET_MODE: ${{ github.event_name == 'workflow_dispatch' && github.event.inputs.paddlefleet_mode || 'develop' }} - PADDLEFLEET_PIN_SHA: ${{ github.event.inputs.paddlefleet_pin_sha }} - PADDLEFLEET_GIT_URL: ${{ github.event.inputs.paddlefleet_git_url }} - PADDLEFLEET_WHEEL_URL: ${{ github.event.inputs.paddlefleet_wheel_url }} - PADDLEFLEET_WHEEL_SHA256: ${{ github.event.inputs.paddlefleet_wheel_sha256 }} - PADDLEFLEET_OPS_WHEEL_URL: ${{ github.event.inputs.paddlefleet_ops_wheel_url }} - PADDLEFLEET_OPS_WHEEL_SHA256: ${{ github.event.inputs.paddlefleet_ops_wheel_sha256 }} defaults: run: @@ -93,76 +55,36 @@ jobs: -e no_proxy \ -e CE_name \ -e python_version \ - -e ALIGNMENT_PADDLEFLEET_MODE \ - -e PADDLEFLEET_PIN_SHA \ - -e PADDLEFLEET_GIT_URL \ - -e PADDLEFLEET_WHEEL_URL \ - -e PADDLEFLEET_WHEEL_SHA256 \ - -e PADDLEFLEET_OPS_WHEEL_URL \ - -e PADDLEFLEET_OPS_WHEEL_SHA256 \ -w /workspace $IMAGE_NAME - name: Checkout Code run: | docker exec -t $container_name /bin/bash -c ' - set -eo pipefail rm -rf * .[^.]* source $work_dir/../../../proxy - command -v timeout >/dev/null || { echo "checkout requires timeout" >&2; exit 127; } - # Keep this inline: the requested repository has not been checked out yet. - # Log labels, never command arguments (which can contain credentials). - checkout_step() { - local label="$1" limit="$2" status=0 started=$SECONDS - shift 2 - echo "checkout begin: ${label} utc=$(date -u +%FT%TZ) timeout=${limit}" - timeout --signal=TERM --kill-after=30s "$limit" "$@" || status=$? - echo "checkout end: ${label} status=${status} elapsed=$((SECONDS - started))s" - if [ "$status" -eq 124 ]; then - echo "checkout timed out: ${label} limit=${limit} status=${status}" >&2 - elif [ "$status" -eq 137 ]; then - echo "checkout terminated: ${label} status=137 (timeout escalation or SIGKILL)" >&2 - elif [ "$status" -ne 0 ]; then - echo "checkout failed: ${label} status=${status}" >&2 - fi - return "$status" - } - if [ "${ALIGNMENT_PADDLEFLEET_MODE:-develop}" = "stack-paired" ]; then - echo "stack-paired: skip CodeSync/develop PaddleFleet.tar; selector runs after python" - else - echo "Download PaddleFleet form https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" - checkout_step paddlefleet-download 600s wget -q --no-proxy --timeout=60 --tries=3 https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate - rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar - cd PaddleFleet - checkout_step paddlefleet-pull 300s git pull - cd - - fi - + echo "Download PaddleFleet form https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" + wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate + rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar + cd PaddleFleet && git pull && cd - + echo "Download ms-swift form https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz" - checkout_step swift-download 600s wget -q --no-proxy --timeout=60 --tries=3 https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate + wget -q --no-proxy https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate rm -rf ms-swift && tar zxf ms-swift.tar.gz && rm -rf ms-swift.tar.gz cd ms-swift git config --global --add safe.directory /workspace/ms-swift - checkout_step swift-pull 300s git pull - checkout_step swift-initial-submodules 900s git submodule update --init --recursive --force - git remote add upstream https://github.com/PFCCLab/ms-swift.git || true + git pull + git submodule update --init --recursive --force if [ -n "$PR_ID" ] && [ "$PR_ID" != "0" ]; then - checkout_step swift-pr-fetch 300s git fetch origin pull/${PR_ID}/head - checkout_step swift-pr-checkout 120s git checkout -B PR_${PR_ID} FETCH_HEAD + git fetch origin pull/${PR_ID}/head + git checkout -b PR_${PR_ID} FETCH_HEAD + git remote add upstream https://github.com/PFCCLab/ms-swift.git echo "Checking out ${BRANCH}..." - checkout_step swift-base-fetch 300s git fetch upstream ${BRANCH}:${BRANCH} || true - checkout_step swift-base-merge 120s git merge ${BRANCH} --no-edit || true + git fetch upstream ${BRANCH}:${BRANCH} + git merge ${BRANCH} --no-edit git diff --numstat ${BRANCH} -- | awk "{print \$NF}" - elif [ -n "$COMMIT_ID" ]; then - echo "workflow_dispatch: checkout COMMIT_ID=${COMMIT_ID} (not BOS main)" - checkout_step swift-fetch-all 900s git fetch --all --tags || true - checkout_step swift-commit-fetch-origin 300s git fetch origin "$COMMIT_ID" || checkout_step swift-commit-fetch-upstream 300s git fetch upstream "$COMMIT_ID" || true - checkout_step swift-commit-checkout 120s git checkout --force "$COMMIT_ID" - checkout_step swift-target-submodules 900s git submodule update --init --recursive --force else - echo "Not in a pull_request event and COMMIT_ID empty. Leaving tarball HEAD." + echo "Not in a pull_request event. Skipping PR-specific operations." fi - echo "checked_out_head=$(git rev-parse HEAD)" - test -z "$COMMIT_ID" || test "$(git rev-parse HEAD)" = "$COMMIT_ID" - checkout_step swift-history 30s git --no-pager log --pretty=oneline -10 + git log --pretty=oneline -10 ' - name: Change python version run: | @@ -182,45 +104,24 @@ jobs: - name: Get Whl run: | docker exec -t $container_name /bin/bash -c ' - set -eo pipefail . /opt/conda/etc/profile.d/conda.sh conda activate py_$python_version python --version ldconfig BOS=https://paddle-github-action.bj.bcebos.com + echo "::group::Download paddlefleet / paddlefleet_ops / megatron-core / mcore-bridge wheels from BOS" cd /workspace - if [ "${ALIGNMENT_PADDLEFLEET_MODE:-develop}" = "stack-paired" ]; then - echo "::group::Select stack-paired PaddleFleet (fail-closed SHA)" - python -m pip install uv - test -x /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh \ - || { echo "::error:: selector missing; checkout did not land COMMIT_ID"; exit 1; } - bash /workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh --dest /workspace \ - || { echo "::error:: selector failed; stop Get Whl (do not download remaining wheels or build ms-swift)"; exit 1; } - bash /workspace/ms-swift/scripts/require_paddlefleet_selector_ok.sh /workspace \ - || { echo "::error:: selector receipt status is not ok; stop Get Whl"; exit 1; } - echo "::endgroup::" - echo "::group::Download remaining wheels from BOS" - for url in \ - $BOS/whl/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl \ - $BOS/whl/mcore_bridge-0.0.0-py3-none-any.whl ; do - echo "Downloading $url" - wget -q --no-proxy --no-check-certificate --tries=3 --timeout=60 "$url" - done - echo "::endgroup::" - else - echo "::group::Download paddlefleet / paddlefleet_ops / megatron-core / mcore-bridge wheels from BOS" - for url in \ - $BOS/PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl \ - $BOS/PaddleFleet/develop/latest/cu130/paddle-release/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl \ - $BOS/whl/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl \ - $BOS/whl/mcore_bridge-0.0.0-py3-none-any.whl ; do - echo "Downloading $url" - wget -q --no-proxy --no-check-certificate --tries=3 --timeout=60 "$url" - done - echo "::endgroup::" - fi + for url in \ + $BOS/PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl \ + $BOS/PaddleFleet/develop/latest/cu130/paddle-release/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl \ + $BOS/whl/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl \ + $BOS/whl/mcore_bridge-0.0.0-py3-none-any.whl ; do + echo "Downloading $url" + wget -q --no-proxy --no-check-certificate --tries=3 --timeout=60 "$url" + done ls -l /workspace/ + echo "::endgroup::" echo "::group::Build ms-swift wheel" cd /workspace/ms-swift @@ -232,7 +133,6 @@ jobs: - name: alignment_model_accuracy run: | docker exec -t $container_name /bin/bash -c ' - set -eo pipefail . /opt/conda/etc/profile.d/conda.sh conda activate py_$python_version python --version @@ -240,24 +140,18 @@ jobs: source $work_dir/../../../proxy export PROXY_URL="${http_proxy}" - ALIGNMENT_PADDLEFLEET_MODE="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" \ - PADDLEFLEET_PIN_SHA="${PADDLEFLEET_PIN_SHA:-}" \ - bash /workspace/ms-swift/scripts/consume_paddlefleet_alignment_pin.sh \ - --env /workspace/paddlefleet_alignment_pin.env \ - --out /workspace/paddlefleet_alignment_pin.consumed.env - set -a - . /workspace/paddlefleet_alignment_pin.consumed.env - set +a + export PADDLEFLEET_WHEEL_PATH="/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" + export PADDLEFLEET_OPS_WHEEL_PATH="/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" export MEGATRON_CORE_WHEEL_PATH=/workspace/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl export MS_SWIFT_WHEEL_PATH=/workspace/upload/ms_swift-0.0.0-py3-none-any.whl export MCORE_BRIDGE_WHEEL_PATH=/workspace/mcore_bridge-0.0.0-py3-none-any.whl - # The BOS wheels are republished under a fixed 0.0.0 filename + # The BOS wheels are republished under a fixed 0.0.0 filename export UV_SKIP_WHEEL_FILENAME_CHECK=1 for whl in "$PADDLEFLEET_WHEEL_PATH" "$PADDLEFLEET_OPS_WHEEL_PATH" \ "$MS_SWIFT_WHEEL_PATH" "$MEGATRON_CORE_WHEEL_PATH" \ "$MCORE_BRIDGE_WHEEL_PATH"; do - [ -e "$whl" ] || { echo "::error:: missing wheel: $whl"; exit 1; } + [ -f "$whl" ] || { echo "::error:: missing wheel: $whl"; exit 1; } echo "using $whl" done python -m pip install uv diff --git a/.github/workflows/citest.yaml b/.github/workflows/citest.yaml index f5ec87ddc1..101e9f3b1a 100644 --- a/.github/workflows/citest.yaml +++ b/.github/workflows/citest.yaml @@ -41,25 +41,15 @@ jobs: # The type of runner that the job will run on runs-on: [self-hosted] timeout-minutes: 240 - env: - # self-hosted paddle-4 glibc < 2.27; runner forces Node 24 for - # actions/checkout@v3 (job 101275701451). Allow Node 20 so checkout runs. - ACTIONS_ALLOW_USE_UNSECURE_NODE_VERSION: true steps: - name: ResetFileMode shell: bash run: | # reset filemode to allow action runner to delete files - # generated by root in docker. Passwordless sudo is not guaranteed - # (self-hosted "sudo: no tty present and no askpass" failed #3). + # generated by root in docker set -e - source ~/.bashrc || true - if command -v sudo >/dev/null 2>&1 && sudo -n true 2>/dev/null; then - sudo -n chown -R "$USER:$USER" "$GITHUB_WORKSPACE" - else - echo "skip sudo chown: no passwordless sudo" - chown -R "$USER:$USER" "$GITHUB_WORKSPACE" 2>/dev/null || true - fi + source ~/.bashrc + sudo chown -R $USER:$USER $GITHUB_WORKSPACE - name: Checkout uses: actions/checkout@v3 diff --git a/scripts/consume_paddlefleet_alignment_pin.sh b/scripts/consume_paddlefleet_alignment_pin.sh deleted file mode 100755 index 1ba0b8e56b..0000000000 --- a/scripts/consume_paddlefleet_alignment_pin.sh +++ /dev/null @@ -1,370 +0,0 @@ -#!/usr/bin/env bash -# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. -# -# Consume selector output in a later docker exec. The caller mode/pin stay -# authoritative: a leftover develop env must not silently downgrade -# stack-paired. A 0.0.0 filename is allowed when the receipt proves an -# explicit URL+sha256 (or a source tree / in-invocation build). Unproven -# develop/latest fallback is refused. - -set -euo pipefail - -usage() { - cat <<'EOF' -Usage: consume_paddlefleet_alignment_pin.sh [--env FILE] [--out FILE] [--self-test] - -Reads paddlefleet_alignment_pin.env + receipt from select_paddlefleet_alignment_pin.sh -and writes a consumed env file for setup_venvs.sh. - -Caller ALIGNMENT_PADDLEFLEET_MODE / PADDLEFLEET_PIN_SHA are the request. -The env file must match that request; it does not override them. -EOF -} - -ENVFILE="${PADDLEFLEET_PIN_ENV:-/workspace/paddlefleet_alignment_pin.env}" -OUTFILE="${PADDLEFLEET_CONSUMED_ENV:-/workspace/paddlefleet_alignment_pin.consumed.env}" -RUN_SELF_TEST=0 -while [[ $# -gt 0 ]]; do - case "$1" in - --env) ENVFILE="${2:?}"; shift 2 ;; - --out) OUTFILE="${2:?}"; shift 2 ;; - --self-test) RUN_SELF_TEST=1; shift ;; - -h|--help) usage; exit 0 ;; - *) echo "unknown arg: $1" >&2; usage; exit 2 ;; - esac -done - -fail() { - echo "::error:: $*" >&2 - exit 1 -} - -parse_envfile() { - FILE_MODE="" - FILE_PIN="" - FILE_SOURCE="" - FILE_RECEIPT="" - FILE_WHEEL="" - FILE_OPS="" - FILE_WHEEL_ORIGIN="" - FILE_OPS_ORIGIN="" - FILE_WHEEL_DIGEST="" - FILE_OPS_DIGEST="" - [[ -f "${ENVFILE}" ]] || return 0 - local line k v - while IFS= read -r line || [[ -n "${line}" ]]; do - [[ -z "${line}" || "${line}" == \#* ]] && continue - k="${line%%=*}" - v="${line#*=}" - case "${k}" in - ALIGNMENT_PADDLEFLEET_MODE) FILE_MODE="${v}" ;; - PADDLEFLEET_PIN_SHA) FILE_PIN="${v}" ;; - PADDLEFLEET_SOURCE_COMMIT) FILE_SOURCE="${v}" ;; - PADDLEFLEET_PIN_RECEIPT) FILE_RECEIPT="${v}" ;; - PADDLEFLEET_WHEEL_PATH) FILE_WHEEL="${v}" ;; - PADDLEFLEET_OPS_WHEEL_PATH) FILE_OPS="${v}" ;; - PADDLEFLEET_WHEEL_ORIGIN) FILE_WHEEL_ORIGIN="${v}" ;; - PADDLEFLEET_OPS_ORIGIN) FILE_OPS_ORIGIN="${v}" ;; - PADDLEFLEET_WHEEL_DIGEST_VERIFIED) FILE_WHEEL_DIGEST="${v}" ;; - PADDLEFLEET_OPS_DIGEST_VERIFIED) FILE_OPS_DIGEST="${v}" ;; - esac - done <"${ENVFILE}" -} - -# Unproven develop fallback: develop_latest origin, or a default 0.0.0 -# filename with no digest proof. A verified ci_metadata/build artifact may -# legally keep the 0.0.0 filename. -unproven_develop_fallback() { - local path="$1" origin="$2" digest="$3" - case "${origin}" in - develop_latest) return 0 ;; - source_tree|build|ci_metadata) - [[ "${origin}" == develop_latest ]] && return 0 - return 1 - ;; - esac - case "${path}" in - */paddlefleet-0.0.0-py3-none-linux_x86_64.whl|*/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl) - [[ "${digest}" == "true" ]] && return 1 - return 0 - ;; - esac - return 1 -} - -check_receipt() { - local receipt="$1" requested_mode="$2" requested_pin="$3" - [[ -f "${receipt}" ]] || fail "stack-paired missing receipt ${receipt}" - python3 - "${receipt}" "${requested_mode}" "${requested_pin}" <<'PY' -import json, sys -path, requested_mode, requested_pin = sys.argv[1:4] -doc = json.load(open(path, encoding="utf-8")) -if doc.get("status") != "ok": - raise SystemExit(f"receipt status={doc.get('status')!r} is not ok") -if requested_mode == "stack-paired": - if doc.get("mode") != "stack-paired": - raise SystemExit( - f"requested stack-paired but receipt mode={doc.get('mode')!r}" - ) - src = doc.get("source") or {} - if requested_pin: - exp = src.get("expected_commit") or "" - act = src.get("actual_commit") or "" - if requested_pin not in (exp, act): - raise SystemExit( - f"receipt source pin mismatch requested={requested_pin} " - f"expected={exp} actual={act}" - ) - if src.get("commit_verified") is not True: - raise SystemExit("receipt source commit_verified is not true") - pairing = (doc.get("pairing") or {}).get("status") - if pairing == "unpaired_default": - raise SystemExit("receipt pairing.status=unpaired_default") - for art in doc.get("artifacts") or []: - origin = art.get("origin") or "" - url = art.get("url") or "" - if origin == "develop_latest": - raise SystemExit(f"artifact {art.get('name')} origin=develop_latest") - if "/develop/latest/" in url or "CodeSync/develop/" in url: - raise SystemExit(f"artifact {art.get('name')} unpaired develop URL") -print("receipt matches request") -PY -} - -write_consumed() { - mkdir -p "$(dirname "${OUTFILE}")" - cat >"${OUTFILE}" <&2 - echo "[paddlefleet-pin-consume] requested_mode=${ALIGNMENT_PADDLEFLEET_MODE} pin=${PADDLEFLEET_PIN_SHA:-} wheel=${PADDLEFLEET_WHEEL_PATH} origin=${PADDLEFLEET_WHEEL_ORIGIN:-}" >&2 -} - -consume() { - local requested_mode="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" - local requested_pin="${PADDLEFLEET_PIN_SHA:-}" - - parse_envfile - - if [[ "${requested_mode}" == "stack-paired" ]]; then - if [[ ! -f "${ENVFILE}" ]]; then - local rec="${ENVFILE%/*}/paddlefleet_alignment_pin_receipt.json" - local rec_status="" - if [[ -f "${rec}" ]]; then - rec_status="$(python3 - "${rec}" <<'PY' -import json, sys -print(json.load(open(sys.argv[1])).get("status") or "") -PY -)" - fi - if [[ "${rec_status}" == "error" ]]; then - fail "stack-paired missing ${ENVFILE} because selector failed (error receipt ${rec}); Get Whl must exit on selector failure. This is not proof that a generated env failed to cross docker exec" - fi - if [[ "${rec_status}" == "ok" ]]; then - fail "stack-paired missing ${ENVFILE}; selector wrote ok receipt but env did not cross docker exec" - fi - fail "stack-paired missing ${ENVFILE} and no selector receipt" - fi - [[ "${FILE_MODE}" == "stack-paired" ]] || fail "requested stack-paired but env mode=${FILE_MODE:-empty} (will not consume a develop leftover)" - if [[ -n "${requested_pin}" ]]; then - local file_id="${FILE_PIN:-${FILE_SOURCE}}" - [[ "${file_id}" == "${requested_pin}" ]] || fail "requested pin ${requested_pin} != env pin/source ${file_id:-empty}" - fi - local receipt="${FILE_RECEIPT:-}" - if [[ -z "${receipt}" && -f "${ENVFILE%/*}/paddlefleet_alignment_pin_receipt.json" ]]; then - receipt="${ENVFILE%/*}/paddlefleet_alignment_pin_receipt.json" - fi - check_receipt "${receipt}" "${requested_mode}" "${requested_pin}" - PADDLEFLEET_WHEEL_PATH="${FILE_WHEEL}" - PADDLEFLEET_OPS_WHEEL_PATH="${FILE_OPS}" - PADDLEFLEET_WHEEL_ORIGIN="${FILE_WHEEL_ORIGIN}" - PADDLEFLEET_OPS_ORIGIN="${FILE_OPS_ORIGIN}" - PADDLEFLEET_PIN_RECEIPT="${receipt}" - PADDLEFLEET_SOURCE_COMMIT="${FILE_SOURCE}" - PADDLEFLEET_PIN_SHA="${requested_pin:-${FILE_PIN}}" - ALIGNMENT_PADDLEFLEET_MODE="stack-paired" - [[ -n "${PADDLEFLEET_WHEEL_PATH}" && -n "${PADDLEFLEET_OPS_WHEEL_PATH}" ]] \ - || fail "stack-paired env missing PADDLEFLEET_WHEEL_PATH or OPS path" - if unproven_develop_fallback "${PADDLEFLEET_WHEEL_PATH}" "${FILE_WHEEL_ORIGIN}" "${FILE_WHEEL_DIGEST}"; then - fail "stack-paired refused unproven develop fallback for paddlefleet: path=${PADDLEFLEET_WHEEL_PATH} origin=${FILE_WHEEL_ORIGIN:-empty} digest_verified=${FILE_WHEEL_DIGEST:-false}" - fi - if unproven_develop_fallback "${PADDLEFLEET_OPS_WHEEL_PATH}" "${FILE_OPS_ORIGIN}" "${FILE_OPS_DIGEST}"; then - fail "stack-paired refused unproven develop fallback for paddlefleet_ops: path=${PADDLEFLEET_OPS_WHEEL_PATH} origin=${FILE_OPS_ORIGIN:-empty} digest_verified=${FILE_OPS_DIGEST:-false}" - fi - else - ALIGNMENT_PADDLEFLEET_MODE="develop" - PADDLEFLEET_WHEEL_PATH="${FILE_WHEEL:-/workspace/paddlefleet-0.0.0-py3-none-linux_x86_64.whl}" - PADDLEFLEET_OPS_WHEEL_PATH="${FILE_OPS:-/workspace/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl}" - PADDLEFLEET_WHEEL_ORIGIN="${FILE_WHEEL_ORIGIN:-develop_latest}" - PADDLEFLEET_OPS_ORIGIN="${FILE_OPS_ORIGIN:-develop_latest}" - PADDLEFLEET_PIN_RECEIPT="${FILE_RECEIPT:-}" - PADDLEFLEET_SOURCE_COMMIT="${FILE_SOURCE:-}" - PADDLEFLEET_PIN_SHA="${requested_pin}" - fi - - if [[ ! -e "${PADDLEFLEET_WHEEL_PATH}" ]]; then - fail "missing paddlefleet path: ${PADDLEFLEET_WHEEL_PATH}" - fi - if [[ ! -e "${PADDLEFLEET_OPS_WHEEL_PATH}" ]]; then - fail "missing paddlefleet_ops path: ${PADDLEFLEET_OPS_WHEEL_PATH}" - fi - write_consumed -} - -write_min_receipt() { - local path="$1" mode="$2" pin="$3" wheel="$4" ops="$5" origin="$6" digest="$7" pairing="$8" - python3 - "${path}" "${mode}" "${pin}" "${wheel}" "${ops}" "${origin}" "${digest}" "${pairing}" <<'PY' -import json, sys -path, mode, pin, wheel, ops, origin, digest, pairing = sys.argv[1:9] -digest_ok = digest == "true" -doc = { - "schema": "paddlefleet-alignment-pin/v1", - "status": "ok", - "mode": mode, - "source": { - "expected_commit": pin or None, - "actual_commit": pin or None, - "commit_verified": bool(pin) and mode == "stack-paired", - }, - "artifacts": [ - {"name": "paddlefleet", "path": wheel, "url": None, "origin": origin, - "digest_verified": digest_ok, "expected_sha256": "abc" if digest_ok else None, - "actual_sha256": "abc" if digest_ok else None}, - {"name": "paddlefleet_ops", "path": ops, "url": None, "origin": origin, - "digest_verified": digest_ok, "expected_sha256": "def" if digest_ok else None, - "actual_sha256": "def" if digest_ok else None}, - ], - "pairing": {"stack_paired_proven": False, "status": pairing, "reason": "fixture"}, -} -open(path, "w", encoding="utf-8").write(json.dumps(doc, indent=2) + "\n") -PY -} - -run_self_test() { - local root self - self="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)/$(basename -- "${BASH_SOURCE[0]}")" - root="$(mktemp -d)" - trap 'rm -rf "${root}"' RETURN - mkdir -p "${root}/PaddleFleet/packages/paddlefleet_ops" - echo tree >"${root}/PaddleFleet/pyproject.toml" - echo ops >"${root}/PaddleFleet/packages/paddlefleet_ops/pyproject.toml" - local pin="aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" - - write_min_receipt "${root}/receipt.json" stack-paired "${pin}" \ - "${root}/PaddleFleet" "${root}/PaddleFleet/packages/paddlefleet_ops" \ - source_tree false source_tree_from_checked_out_pin - cat >"${root}/pin.env" <"${root}/devwhl/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" - echo dummy >"${root}/devwhl/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" - write_min_receipt "${root}/develop-receipt.json" develop "" \ - "${root}/devwhl/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" \ - "${root}/devwhl/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" \ - develop_latest false unpaired_default - cat >"${root}/develop.env" <&2 - exit 1 - fi - - # 2) verified URL+hash artifact may keep the 0.0.0 filename. - write_min_receipt "${root}/named-receipt.json" stack-paired "${pin}" \ - "${root}/devwhl/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" \ - "${root}/devwhl/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" \ - ci_metadata true unproven - cat >"${root}/named.env" <"${root}/bare.env" <&2 - exit 1 - fi - - # Selector clone/fail: error receipt, no env. Missing env is the - # consequence, not proof a generated env failed to cross docker exec. - mkdir -p "${root}/sel-fail" - cat >"${root}/sel-fail/paddlefleet_alignment_pin_receipt.json" <<'EOF' -{"schema":"paddlefleet-alignment-pin/v1","status":"error","detail":"git clone failed: github.com:443","mode":"stack-paired"} -EOF - if ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${pin}" \ - bash "${self}" --env "${root}/sel-fail/paddlefleet_alignment_pin.env" \ - --out "${root}/sel-fail.consumed.env" 2>"${root}/sel-fail.err"; then - echo "self-test FAIL: selector error receipt was consumed" >&2 - exit 1 - fi - grep -q "because selector failed" "${root}/sel-fail.err" - if grep -q "selector wrote ok receipt but env did not cross docker exec" "${root}/sel-fail.err"; then - echo "self-test FAIL: selector error misclassified as env-handoff" >&2 - exit 1 - fi - - echo "consume_paddlefleet_alignment_pin self-test OK" -} - -if [[ "${RUN_SELF_TEST}" == 1 ]]; then - run_self_test - exit 0 -fi -consume diff --git a/scripts/dependence/build.sh b/scripts/dependence/build.sh index 6b9129513c..de73702692 100644 --- a/scripts/dependence/build.sh +++ b/scripts/dependence/build.sh @@ -79,3 +79,4 @@ echo -e "\033[32m ---- make ms-swift.tar.gz \033[0m" swift_tar echo -e "\033[32m ---- build ms-swift whl \033[0m" swift_build + diff --git a/scripts/require_paddlefleet_selector_ok.sh b/scripts/require_paddlefleet_selector_ok.sh deleted file mode 100755 index f34221204a..0000000000 --- a/scripts/require_paddlefleet_selector_ok.sh +++ /dev/null @@ -1,40 +0,0 @@ -#!/usr/bin/env bash -# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. -# -# Called from Get Whl after select_paddlefleet_alignment_pin.sh. -# Keep this file free of nested quotes so the workflow docker exec -# single-quoted -c script can invoke it without breaking the host shell. -# A selector error receipt is not an env-handoff failure. - -set -euo pipefail - -DEST="${1:-/workspace}" -REC="${DEST}/paddlefleet_alignment_pin_receipt.json" -ENVF="${DEST}/paddlefleet_alignment_pin.env" - -if [[ ! -f "${REC}" ]]; then - echo "::error:: missing selector receipt ${REC}" >&2 - exit 1 -fi - -python3 - "${REC}" <<'PY' -import json -import sys - -path = sys.argv[1] -doc = json.load(open(path, encoding="utf-8")) -status = doc.get("status") -if status != "ok": - raise SystemExit( - f"selector receipt status={status!r} is not ok; " - "Get Whl must stop (do not download remaining wheels or build)" - ) -print("selector receipt status=ok") -PY - -if [[ ! -f "${ENVF}" ]]; then - echo "::error:: selector did not write ${ENVF}" >&2 - exit 1 -fi - -cat "${REC}" diff --git a/scripts/select_paddlefleet_alignment_pin.sh b/scripts/select_paddlefleet_alignment_pin.sh deleted file mode 100755 index cfa778fa9d..0000000000 --- a/scripts/select_paddlefleet_alignment_pin.sh +++ /dev/null @@ -1,889 +0,0 @@ -#!/usr/bin/env bash -# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -# Select PaddleFleet source + wheels for alignment_model_accuracy. -# -# Default (ALIGNMENT_PADDLEFLEET_MODE=develop or unset): -# historical CodeSync/develop tarball + BOS develop/latest wheels. -# Cases are not filtered. -# -# Explicit (ALIGNMENT_PADDLEFLEET_MODE=stack-paired): fail-closed. -# Fetch PADDLEFLEET_PIN_SHA at depth 1 (not a full default-branch clone). -# Transient git RPC/curl failures retry up to 3 clean dests; other -# fetch errors fail closed. git rev-parse HEAD must equal the pin. -# Artifacts: caller URL+sha256 (from Build Fleet whl / Actions metadata) -# or build from the checked-out tree. Independent digest matches do not -# prove the wheels were produced from that commit — receipt records -# source_commit vs artifact sha256 separately and pairing as unproven -# unless this invocation built the files from the pin. -# git/download/checkout failures still write an error receipt. - -set -euo pipefail - -usage() { - cat <<'EOF' -Usage: select_paddlefleet_alignment_pin.sh [--dest DIR] [--self-test] - ---self-test ignores --dest and uses an offline fixture (no network). - -Env: - ALIGNMENT_PADDLEFLEET_MODE develop (default) | stack-paired - PADDLEFLEET_PIN_SHA required 40-hex commit in stack-paired - PADDLEFLEET_GIT_URL git remote or local repo (stack-paired) - PADDLEFLEET_WHEEL_URL optional explicit wheel (https or local path) - PADDLEFLEET_WHEEL_SHA256 required with WHEEL_URL - PADDLEFLEET_OPS_WHEEL_URL optional explicit ops wheel - PADDLEFLEET_OPS_WHEEL_SHA256 required with OPS URL - PADDLEFLEET_BUILD_CMD optional; default uv build paddlefleet - PADDLEFLEET_BUILD_OPS_CMD optional; default uv build paddlefleet-ops - ALIGNMENT_PADDLEFLEET_DEST output directory (default /workspace) -EOF -} - -MODE="${ALIGNMENT_PADDLEFLEET_MODE:-develop}" -DEST="${ALIGNMENT_PADDLEFLEET_DEST:-/workspace}" -RUN_SELF_TEST=0 -while [[ $# -gt 0 ]]; do - case "$1" in - --dest) - DEST="${2:?--dest requires a path}" - shift 2 - ;; - --self-test) - RUN_SELF_TEST=1 - shift - ;; - -h|--help) - usage - exit 0 - ;; - *) - echo "unknown arg: $1" >&2 - usage - exit 2 - ;; - esac -done - -BOS="${PADDLEFLEET_BOS:-https://paddle-github-action.bj.bcebos.com}" -DEFAULT_TAR_URL="https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar" -DEFAULT_WHL_URL="${BOS}/PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" -DEFAULT_OPS_URL="${BOS}/PaddleFleet/develop/latest/cu130/paddle-release/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" -GIT_URL="${PADDLEFLEET_GIT_URL:-https://github.com/PaddlePaddle/PaddleFleet.git}" -PIN_SHA="${PADDLEFLEET_PIN_SHA:-}" -WHEEL_URL="${PADDLEFLEET_WHEEL_URL:-}" -WHEEL_SHA="${PADDLEFLEET_WHEEL_SHA256:-}" -OPS_URL="${PADDLEFLEET_OPS_WHEEL_URL:-}" -OPS_SHA="${PADDLEFLEET_OPS_WHEEL_SHA256:-}" -BUILD_CMD="${PADDLEFLEET_BUILD_CMD:-}" -BUILD_OPS_CMD="${PADDLEFLEET_BUILD_OPS_CMD:-}" - -ACTUAL_SHA="" -SOURCE_VERIFIED=false -PADDLEFLEET_WHEEL_PATH="" -PADDLEFLEET_OPS_WHEEL_PATH="" -ACTUAL_WHEEL_SHA="" -ACTUAL_OPS_SHA="" -WHEEL_DIGEST_VERIFIED=false -OPS_DIGEST_VERIFIED=false -WHEEL_ORIGIN="" -OPS_ORIGIN="" -WHEEL_BUILT_FROM_COMMIT="" -OPS_BUILT_FROM_COMMIT="" -LOADED_FROM="" -RECEIPT_WRITTEN=0 - -log() { echo "[paddlefleet-pin] $*" >&2; } - -sha256_file() { sha256sum -- "$1" | awk '{print $1}'; } - -unpaired_url() { - case "$1" in - *"/develop/latest/"*|*"CodeSync/develop/"*) return 0 ;; - *) return 1 ;; - esac -} - -pairing_fields() { - local proven=false - local status="unproven" - local reason="wheel/ops digest match does not prove production from source_commit" - if [[ "${MODE}" == "develop" ]]; then - status="unpaired_default" - reason="develop tarball and develop/latest wheels; not a stack pin" - elif [[ "${WHEEL_ORIGIN}" == "source_tree" && "${OPS_ORIGIN}" == "source_tree" \ - && "${SOURCE_VERIFIED}" == "true" ]]; then - status="source_tree_from_checked_out_pin" - reason="this invocation exported checked-out source trees; not a wheel digest proof" - elif [[ "${WHEEL_ORIGIN}" == "build" && "${OPS_ORIGIN}" == "build" \ - && "${WHEEL_BUILT_FROM_COMMIT}" == "${ACTUAL_SHA}" \ - && "${OPS_BUILT_FROM_COMMIT}" == "${ACTUAL_SHA}" \ - && "${SOURCE_VERIFIED}" == "true" ]]; then - status="built_from_checked_out_pin" - reason="this invocation built both artifacts from checked-out source_commit; not a remote-stack proof" - fi - printf '%s\t%s\t%s\n' "${proven}" "${status}" "${reason}" -} - -write_receipt() { - local status="$1" detail="${2:-}" - mkdir -p "${DEST}" - local receipt="${DEST}/paddlefleet_alignment_pin_receipt.json" - local pair - pair="$(pairing_fields)" - local stack_proven pairing_status pairing_reason - stack_proven="${pair%%$'\t'*}" - pair="${pair#*$'\t'}" - pairing_status="${pair%%$'\t'*}" - pairing_reason="${pair#*$'\t'}" - if ! command -v python3 >/dev/null 2>&1; then - printf '{"schema":"paddlefleet-alignment-pin/v1","status":"%s","detail":"%s"}\n' \ - "${status}" "${detail}" >"${receipt}" - RECEIPT_WRITTEN=1 - return 0 - fi - python3 - "${receipt}" "${status}" "${detail}" "${stack_proven}" \ - "${pairing_status}" "${pairing_reason}" <<'PY' -import json, os, sys -from datetime import datetime, timezone -path, status, detail, stack_proven, pairing_status, pairing_reason = sys.argv[1:7] - -def art(name, pth, url, exp, act, digest_ok, origin, built_from): - if not pth: - return None - return { - "name": name, - "path": pth, - "url": url or None, - "expected_sha256": exp or None, - "actual_sha256": act or None, - "digest_verified": digest_ok == "true", - "origin": origin or None, - "built_from_commit": built_from or None, - } - -arts = [a for a in ( - art("paddlefleet", os.environ.get("PADDLEFLEET_WHEEL_PATH", ""), - os.environ.get("WHEEL_URL", ""), os.environ.get("WHEEL_SHA", ""), - os.environ.get("ACTUAL_WHEEL_SHA", ""), os.environ.get("WHEEL_DIGEST_VERIFIED", "false"), - os.environ.get("WHEEL_ORIGIN", ""), os.environ.get("WHEEL_BUILT_FROM_COMMIT", "")), - art("paddlefleet_ops", os.environ.get("PADDLEFLEET_OPS_WHEEL_PATH", ""), - os.environ.get("OPS_URL", ""), os.environ.get("OPS_SHA", ""), - os.environ.get("ACTUAL_OPS_SHA", ""), os.environ.get("OPS_DIGEST_VERIFIED", "false"), - os.environ.get("OPS_ORIGIN", ""), os.environ.get("OPS_BUILT_FROM_COMMIT", "")), -) if a] - -doc = { - "schema": "paddlefleet-alignment-pin/v1", - "status": status, - "detail": detail, - "captured_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), - "mode": os.environ.get("MODE"), - "dest": os.environ.get("DEST"), - "source": { - "git_url": os.environ.get("GIT_URL") or None, - "expected_commit": os.environ.get("PIN_SHA") or None, - "actual_commit": os.environ.get("ACTUAL_SHA") or None, - "commit_verified": os.environ.get("SOURCE_VERIFIED") == "true", - }, - "loaded_from": os.environ.get("LOADED_FROM") or None, - "artifacts": arts, - "pairing": { - "stack_paired_proven": stack_proven == "true", - "status": pairing_status, - "reason": pairing_reason, - }, - "default_urls": { - "source_tar": os.environ.get("DEFAULT_TAR_URL"), - "paddlefleet_wheel": os.environ.get("DEFAULT_WHL_URL"), - "paddlefleet_ops_wheel": os.environ.get("DEFAULT_OPS_URL"), - }, - "cases_preserved": ["MinimaxV2.5_EP2", "GLM45Air_EP2"], - "unpaired_develop_rejected_in_stack_paired": True, -} -open(path, "w", encoding="utf-8").write(json.dumps(doc, indent=2) + "\n") -print("[paddlefleet-pin] receipt", path, file=sys.stderr) -PY - RECEIPT_WRITTEN=1 -} - -export_receipt_env() { - export MODE DEST GIT_URL PIN_SHA ACTUAL_SHA SOURCE_VERIFIED LOADED_FROM - export PADDLEFLEET_WHEEL_PATH PADDLEFLEET_OPS_WHEEL_PATH - export WHEEL_URL WHEEL_SHA ACTUAL_WHEEL_SHA WHEEL_DIGEST_VERIFIED WHEEL_ORIGIN WHEEL_BUILT_FROM_COMMIT - export OPS_URL OPS_SHA ACTUAL_OPS_SHA OPS_DIGEST_VERIFIED OPS_ORIGIN OPS_BUILT_FROM_COMMIT - export DEFAULT_TAR_URL DEFAULT_WHL_URL DEFAULT_OPS_URL -} - -fail() { - trap - ERR - local msg="$1" - log "FAIL: ${msg}" - export_receipt_env - write_receipt "error" "${msg}" - echo "::error:: ${msg}" >&2 - exit 1 -} - -on_err() { - local rc=$? - if [[ "${RECEIPT_WRITTEN}" == 1 || "${RUN_SELF_TEST}" == 1 ]]; then - return "${rc}" - fi - fail "command failed rc=${rc}" -} -trap 'on_err' ERR - -write_envfile() { - cat >"${DEST}/paddlefleet_alignment_pin.env" < ${out}" - mkdir -p "$(dirname "${out}")" - if [[ "${url}" == file://* ]]; then - local src="${url#file://}" - [[ -f "${src}" ]] || fail "download failed, local file missing: ${src}" - cp -- "${src}" "${out}" || fail "download copy failed: ${src}" - return 0 - fi - if [[ "${url}" == /* ]]; then - [[ -f "${url}" ]] || fail "download failed, local file missing: ${url}" - cp -- "${url}" "${out}" || fail "download copy failed: ${url}" - return 0 - fi - if wget -q --no-proxy --no-check-certificate --tries=2 --timeout=15 -O "${out}" "${url}"; then - return 0 - fi - fail "download failed: ${url}" -} - -# Must not run inside $(); fail() has to exit this shell. -require_digest() { - local path="$1" expected="$2" label="$3" actual="$4" - [[ -f "${path}" ]] || fail "missing ${label}: ${path}" - [[ -n "${expected}" ]] || fail "stack-paired missing ${label} sha256" - if [[ "${actual}" != "${expected}" ]]; then - fail "stack-paired ${label} sha256 mismatch expected=${expected} actual=${actual}" - fi -} - -# Auth / missing-object / missing-repo are permanent. Do not treat a -# generic "RPC failed" as transient: HTTP 401/403 also say RPC failed. -_is_permanent_git_err() { - case "$1" in - *"HTTP 401"*|*"HTTP 403"*|*"Authentication failed"*|*"access denied"*|*"Access denied"*|*"Permission denied"*|*"not our ref"*|*"does not appear to be a git repository"*|*"Could not read from remote repository"*|*"remote: Write access"*) - return 0 - ;; - esac - return 1 -} - -_is_transient_git_err() { - _is_permanent_git_err "$1" && return 1 - case "$1" in - *"curl 56"*|*"Connection timed out"*|*"Couldn't connect to server"*|*"Failed to connect to github.com port 443"*|*"early EOF"*|*"fetch-pack: unexpected disconnect"*|*"bytes of body are still expected"*|*"invalid index-pack output"*|*"Connection reset by peer"*) - return 0 - ;; - esac - return 1 -} - -checkout_pin() { - [[ "${PIN_SHA}" =~ ^[0-9a-fA-F]{40}$ ]] || fail "stack-paired requires PADDLEFLEET_PIN_SHA (40 hex), got '${PIN_SHA}'" - PIN_SHA="$(printf '%s' "${PIN_SHA}" | tr 'A-F' 'a-f')" - local dest="${DEST}/PaddleFleet" - local attempt max_attempts=3 - local last_err="" - local fetch_err="${DEST}/.git-fetch.err" - for attempt in $(seq 1 "${max_attempts}"); do - rm -rf "${dest}" - log "fetch ${GIT_URL} ${PIN_SHA} (attempt ${attempt}/${max_attempts})" - if ! git init --quiet "${dest}" >/dev/null 2>"${DEST}/.git-init.err"; then - fail "git init failed: $(tr '\n' ' ' <"${DEST}/.git-init.err")" - fi - if ! git -C "${dest}" remote add origin "${GIT_URL}" >/dev/null 2>"${DEST}/.git-remote.err"; then - fail "git remote add failed: $(tr '\n' ' ' <"${DEST}/.git-remote.err")" - fi - git -C "${dest}" config advice.detachedHead false || true - if GIT_TERMINAL_PROMPT=0 git -C "${dest}" fetch --depth=1 --no-tags origin "${PIN_SHA}" \ - >/dev/null 2>"${fetch_err}"; then - if ! git -C "${dest}" checkout --quiet --force --detach "${PIN_SHA}" >/dev/null 2>"${DEST}/.git-co.err"; then - fail "git checkout failed for ${PIN_SHA}: $(tr '\n' ' ' <"${DEST}/.git-co.err")" - fi - ACTUAL_SHA="$(git -C "${dest}" rev-parse HEAD)" - if [[ "${ACTUAL_SHA}" != "${PIN_SHA}" ]]; then - SOURCE_VERIFIED=false - fail "stack-paired source SHA mismatch expected=${PIN_SHA} actual=${ACTUAL_SHA}" - fi - SOURCE_VERIFIED=true - log "source commit verified ${ACTUAL_SHA}" - return 0 - fi - last_err="$(tr '\n' ' ' <"${fetch_err}")" - if ! _is_transient_git_err "${last_err}"; then - fail "git fetch failed: ${last_err}" - fi - log "transient fetch failure attempt ${attempt}/${max_attempts}: ${last_err}" - done - fail "git fetch failed: ${last_err}" -} - -# Sets DEST_PATH and DEST_SHA in the caller. Must run in this shell so -# fail() writes the receipt (never wrap this in $()). -acquire_explicit() { - local url="$1" expected="$2" dest_name="$3" label="$4" - unpaired_url "${url}" && fail "stack-paired rejects unpaired ${label} URL: ${url}" - [[ -n "${expected}" ]] || fail "stack-paired ${label} URL requires matching sha256" - download "${url}" "${DEST}/${dest_name}" - DEST_PATH="${DEST}/${dest_name}" - DEST_SHA="$(sha256_file "${DEST_PATH}")" - # Record path/digest before require_digest so a mismatch receipt still has them. - if [[ "${label}" == paddlefleet\ wheel ]]; then - PADDLEFLEET_WHEEL_PATH="${DEST_PATH}" - ACTUAL_WHEEL_SHA="${DEST_SHA}" - WHEEL_ORIGIN="ci_metadata" - else - PADDLEFLEET_OPS_WHEEL_PATH="${DEST_PATH}" - ACTUAL_OPS_SHA="${DEST_SHA}" - OPS_ORIGIN="ci_metadata" - fi - require_digest "${DEST_PATH}" "${expected}" "${label}" "${DEST_SHA}" - LOADED_FROM="${LOADED_FROM:+${LOADED_FROM};}${DEST_PATH} from ${url}" -} - -run_build() { - local cmd="$1" glob="$2" label="$3" - mkdir -p "${DEST}/dist" - log "build ${label}: ${cmd}" - if ! (cd "${DEST}/PaddleFleet" && bash -lc "${cmd}"); then - fail "stack-paired build failed for ${label}" - fi - local built - built="$(ls -1 ${glob} 2>/dev/null | head -n 1 || true)" - [[ -n "${built}" && -f "${built}" ]] || fail "stack-paired build produced no ${label} (glob ${glob})" - local dest_name - dest_name="$(basename "${built}")" - cp -f -- "${built}" "${DEST}/${dest_name}" - DEST_PATH="${DEST}/${dest_name}" - DEST_SHA="$(sha256_file "${DEST_PATH}")" - LOADED_FROM="${LOADED_FROM:+${LOADED_FROM};}built ${DEST_PATH} from ${ACTUAL_SHA}" -} - -acquire_wheel() { - if [[ -n "${WHEEL_URL}" ]]; then - acquire_explicit "${WHEEL_URL}" "${WHEEL_SHA}" "paddlefleet.whl" "paddlefleet wheel" - PADDLEFLEET_WHEEL_PATH="${DEST_PATH}" - ACTUAL_WHEEL_SHA="${DEST_SHA}" - WHEEL_DIGEST_VERIFIED=true - WHEEL_ORIGIN="ci_metadata" - WHEEL_BUILT_FROM_COMMIT="" - else - local cmd="${BUILD_CMD:-uv build --wheel --package paddlefleet --out-dir '${DEST}/dist' --clear}" - run_build "${cmd}" "${DEST}/dist/paddlefleet-*.whl" "paddlefleet wheel" - PADDLEFLEET_WHEEL_PATH="${DEST_PATH}" - ACTUAL_WHEEL_SHA="${DEST_SHA}" - WHEEL_DIGEST_VERIFIED=true - WHEEL_ORIGIN="build" - WHEEL_BUILT_FROM_COMMIT="${ACTUAL_SHA}" - fi -} - -acquire_ops() { - if [[ -n "${OPS_URL}" ]]; then - acquire_explicit "${OPS_URL}" "${OPS_SHA}" "paddlefleet_ops.whl" "paddlefleet_ops wheel" - PADDLEFLEET_OPS_WHEEL_PATH="${DEST_PATH}" - ACTUAL_OPS_SHA="${DEST_SHA}" - OPS_DIGEST_VERIFIED=true - OPS_ORIGIN="ci_metadata" - OPS_BUILT_FROM_COMMIT="" - else - local cmd="${BUILD_OPS_CMD:-uv build --wheel --package paddlefleet-ops --out-dir '${DEST}/dist' --no-build-isolation}" - run_build "${cmd}" "${DEST}/dist/paddlefleet_ops-*.whl" "paddlefleet_ops wheel" - PADDLEFLEET_OPS_WHEEL_PATH="${DEST_PATH}" - ACTUAL_OPS_SHA="${DEST_SHA}" - OPS_DIGEST_VERIFIED=true - OPS_ORIGIN="build" - OPS_BUILT_FROM_COMMIT="${ACTUAL_SHA}" - fi -} - -fetch_default() { - log "mode=develop (historical unpaired CodeSync tarball)" - download "${DEFAULT_TAR_URL}" "${DEST}/PaddleFleet.tar" - rm -rf "${DEST}/PaddleFleet" - tar xf "${DEST}/PaddleFleet.tar" -C "${DEST}" - rm -f "${DEST}/PaddleFleet.tar" - if [[ -d "${DEST}/PaddleFleet/.git" ]]; then - git -C "${DEST}/PaddleFleet" pull || log "git pull skipped" - ACTUAL_SHA="$(git -C "${DEST}/PaddleFleet" rev-parse HEAD 2>/dev/null || true)" - fi - download "${DEFAULT_WHL_URL}" "${DEST}/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" - download "${DEFAULT_OPS_URL}" "${DEST}/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" - PADDLEFLEET_WHEEL_PATH="${DEST}/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" - PADDLEFLEET_OPS_WHEEL_PATH="${DEST}/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" - WHEEL_URL="${DEFAULT_WHL_URL}" - OPS_URL="${DEFAULT_OPS_URL}" - ACTUAL_WHEEL_SHA="$(sha256_file "${PADDLEFLEET_WHEEL_PATH}")" - ACTUAL_OPS_SHA="$(sha256_file "${PADDLEFLEET_OPS_WHEEL_PATH}")" - WHEEL_ORIGIN="develop_latest" - OPS_ORIGIN="develop_latest" - SOURCE_VERIFIED=false - WHEEL_DIGEST_VERIFIED=false - OPS_DIGEST_VERIFIED=false - LOADED_FROM="develop_latest ${DEFAULT_WHL_URL} ${DEFAULT_OPS_URL}" - export_receipt_env - write_envfile - write_receipt "ok" "develop tarball and develop/latest wheels; unpaired with a stack pin" -} - -acquire_source_tree() { - local src="${DEST}/PaddleFleet" - local ops="${src}/packages/paddlefleet_ops" - [[ -d "${src}" ]] || fail "stack-paired source tree missing: ${src}" - [[ -f "${src}/pyproject.toml" ]] || fail "stack-paired source tree missing pyproject.toml: ${src}" - [[ -d "${ops}" ]] || fail "stack-paired ops source tree missing: ${ops}" - PADDLEFLEET_WHEEL_PATH="${src}" - PADDLEFLEET_OPS_WHEEL_PATH="${ops}" - WHEEL_ORIGIN="source_tree" - OPS_ORIGIN="source_tree" - WHEEL_BUILT_FROM_COMMIT="${ACTUAL_SHA}" - OPS_BUILT_FROM_COMMIT="${ACTUAL_SHA}" - WHEEL_DIGEST_VERIFIED=false - OPS_DIGEST_VERIFIED=false - LOADED_FROM="source_tree ${src} ${ops} from ${ACTUAL_SHA}" - log "source-tree paths ${src} ${ops}" -} - -fetch_stack_paired() { - log "mode=stack-paired" - checkout_pin - if [[ -n "${WHEEL_URL}" || -n "${OPS_URL}" || -n "${BUILD_CMD}" || -n "${BUILD_OPS_CMD}" ]]; then - acquire_wheel - acquire_ops - else - # No CI wheel URL and no explicit build: export the checked-out trees. - # A later docker exec must source paddlefleet_alignment_pin.env. - acquire_source_tree - fi - export_receipt_env - write_envfile - local pair pairing_status - pair="$(pairing_fields)" - pair="${pair#*$'\t'}" - pairing_status="${pair%%$'\t'*}" - write_receipt "ok" "source_commit checked out; pairing.status=${pairing_status}; stack_paired_proven=false" -} - -install_offline_stubs() { - local bin="$1" - mkdir -p "${bin}" - cat >"${bin}/wget" <<'WGET' -#!/usr/bin/env bash -out="" -url="" -while [[ $# -gt 0 ]]; do - case "$1" in - -O) out="$2"; shift 2 ;; - --*) shift ;; - *) url="$1"; shift ;; - esac -done -if [[ -z "${url}" || -z "${out}" ]]; then - echo "wget-stub: missing url/out" >&2 - exit 1 -fi -if [[ "${url}" == http://* || "${url}" == https://* ]]; then - echo "wget-stub: blocked network ${url}" >&2 - exit 1 -fi -src="${url#file://}" -if [[ -f "${src}" ]]; then - cp -- "${src}" "${out}" - exit 0 -fi -echo "wget-stub: not a local file ${url}" >&2 -exit 1 -WGET - chmod +x "${bin}/wget" -} - -run_self_test() { - trap - ERR - local root script - root="$(mktemp -d)" - script="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)/$(basename -- "${BASH_SOURCE[0]}")" - trap 'rm -rf "${root}"' RETURN - install_offline_stubs "${root}/bin" - export PATH="${root}/bin:${PATH}" - - git init -q "${root}/upstream" - git -C "${root}/upstream" config user.email test@example.com - git -C "${root}/upstream" config user.name test - echo source-a >"${root}/upstream/README" - mkdir -p "${root}/upstream/packages/paddlefleet_ops" - printf '%s\n' '[project]' 'name = "paddlefleet"' >"${root}/upstream/pyproject.toml" - printf '%s\n' '[project]' 'name = "paddlefleet-ops"' >"${root}/upstream/packages/paddlefleet_ops/pyproject.toml" - git -C "${root}/upstream" add README pyproject.toml packages - git -C "${root}/upstream" commit -q -m a - local sha_a sha_b - sha_a="$(git -C "${root}/upstream" rev-parse HEAD)" - echo source-b >"${root}/upstream/README" - git -C "${root}/upstream" add README - git -C "${root}/upstream" commit -q -m b - sha_b="$(git -C "${root}/upstream" rev-parse HEAD)" - - mkdir -p "${root}/art" - echo py-body >"${root}/art/py.whl" - echo ops-body >"${root}/art/ops.whl" - local py_sha ops_sha - py_sha="$(sha256_file "${root}/art/py.whl")" - ops_sha="$(sha256_file "${root}/art/ops.whl")" - - expect_fail() { - local dest="$1" - local needle="$2" - shift 2 - mkdir -p "${dest}" - if "$@"; then - echo "self-test FAIL: expected failure (${needle})" >&2 - exit 1 - fi - local rec="${dest}/paddlefleet_alignment_pin_receipt.json" - [[ -f "${rec}" ]] || { echo "self-test FAIL: missing error receipt ${rec}" >&2; exit 1; } - grep -q '"status": "error"' "${rec}" - grep -q "${needle}" "${rec}" - echo "[self-test] fail-closed ${dest}: ${needle}" - } - - local run - run() { env PATH="${root}/bin:${PATH}" "$@"; } - - expect_fail "${root}/m1" "PADDLEFLEET_PIN_SHA" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA= \ - bash "${script}" --dest "${root}/m1" - - expect_fail "${root}/m2" "rejects unpaired" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - PADDLEFLEET_WHEEL_URL="${DEFAULT_WHL_URL}" \ - PADDLEFLEET_WHEEL_SHA256="${py_sha}" \ - PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ - PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ - bash "${script}" --dest "${root}/m2" - - expect_fail "${root}/m3" "git fetch failed" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="0000000000000000000000000000000000000000" \ - PADDLEFLEET_GIT_URL="${root}/upstream" \ - bash "${script}" --dest "${root}/m3" - - # Real checksum mismatch after a successful local copy (not a wget miss). - expect_fail "${root}/m4" "sha256 mismatch" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - PADDLEFLEET_WHEEL_URL="${root}/art/py.whl" \ - PADDLEFLEET_WHEEL_SHA256="deadbeefdeadbeefdeadbeefdeadbeefdeadbeefdeadbeefdeadbeefdeadbeef" \ - PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ - PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ - bash "${script}" --dest "${root}/m4" - python3 - "${root}/m4/paddlefleet_alignment_pin_receipt.json" "${py_sha}" <<'PY' -import json, sys -doc = json.load(open(sys.argv[1])) -assert doc["status"] == "error" -wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") -assert wheel["actual_sha256"] == sys.argv[2] -assert wheel["digest_verified"] is False -assert wheel["actual_sha256"] != (wheel.get("expected_sha256") or "") -print("m4 checksum-mismatch receipt has actual digest, not a download miss") -PY - - expect_fail "${root}/m5" "produced no paddlefleet" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - PADDLEFLEET_BUILD_CMD="mkdir -p '${root}/m5/dist'" \ - PADDLEFLEET_BUILD_OPS_CMD="true" \ - bash "${script}" --dest "${root}/m5" - - expect_fail "${root}/m6" "download failed" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - PADDLEFLEET_WHEEL_URL="https://example.invalid/paddlefleet.whl" \ - PADDLEFLEET_WHEEL_SHA256="${py_sha}" \ - PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ - PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ - bash "${script}" --dest "${root}/m6" - - expect_fail "${root}/m7" "git fetch failed" \ - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" \ - PADDLEFLEET_GIT_URL="${root}/no-such-remote" \ - bash "${script}" --dest "${root}/m7" - - assert_error_unverified() { - python3 - "$1" <<'PY' -import json, sys -doc = json.load(open(sys.argv[1])) -assert doc["status"] == "error", doc -assert doc["source"]["commit_verified"] is False, doc["source"] -print("error receipt commit_verified=false") -PY - } - - assert_pin_checkout() { - local repo="$1" sha="$2" - [[ "$(git -C "${repo}" rev-parse HEAD)" == "${sha}" ]] - [[ -f "${repo}/.git/shallow" ]] || { echo "self-test FAIL: missing ${repo}/.git/shallow" >&2; exit 1; } - [[ "$(git -C "${repo}" rev-list --count HEAD)" == 1 ]] || { - echo "self-test FAIL: ${repo} rev-list count != 1 (not depth=1)" >&2 - exit 1 - } - if git -C "${repo}" symbolic-ref -q HEAD >/dev/null; then - echo "self-test FAIL: ${repo} HEAD is a branch, not detached pin" >&2 - git -C "${repo}" symbolic-ref HEAD >&2 - exit 1 - fi - } - - write_git_wrapper() { - local bindir="$1" mode="$2" - local real_git - real_git="$(command -v git)" - mkdir -p "${bindir}" - cat >"${bindir}/git" <>"\$STALE" - exit 99 - fi - if [[ "\$has_depth" -ne 1 || "\$has_no_tags" -ne 1 ]]; then - echo "fetch missing --depth=1/--no-tags: \${args[*]}" >>"\$STALE" - exit 99 - fi - count=\$((count + 1)) - printf '%s\n' "\$count" >"\$STATE" - case "${mode}" in - 401) - echo "RPC failed; HTTP 401 curl 22 The requested URL returned error: 401" >&2 - echo "fatal: Authentication failed" >&2 - printf 'sentinel\n' >"\$workdir/.retry-sentinel" - exit 128 - ;; - exhaust) - echo "error: RPC failed; curl 56 Recv failure: Connection timed out" >&2 - echo "error: 9515 bytes of body are still expected" >&2 - echo "fatal: early EOF" >&2 - printf 'sentinel\n' >"\$workdir/.retry-sentinel" - exit 128 - ;; - retry) - if [[ "\$count" -lt 3 ]]; then - echo "error: RPC failed; curl 56 Recv failure: Connection timed out" >&2 - echo "error: 9515 bytes of body are still expected" >&2 - echo "fetch-pack: unexpected disconnect while reading sideband packet" >&2 - echo "fatal: early EOF" >&2 - echo "fatal: fetch-pack: invalid index-pack output" >&2 - printf 'sentinel\n' >"\$workdir/.retry-sentinel" - exit 128 - fi - ;; - esac -fi -exec "\$real" "\$@" -GITWRAP - chmod +x "${bindir}/git" - } - - assert_error_unverified "${root}/m3/paddlefleet_alignment_pin_receipt.json" - assert_error_unverified "${root}/m7/paddlefleet_alignment_pin_receipt.json" - - # Permanent HTTP 401 (also says RPC failed) must not retry. - write_git_wrapper "${root}/bin-401" 401 - expect_fail "${root}/m8" "git fetch failed" \ - env PATH="${root}/bin-401:${root}/bin:${PATH}" \ - ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - bash "${script}" --dest "${root}/m8" - [[ "$(cat "${root}/bin-401/count")" == 1 ]] - [[ ! -f "${root}/bin-401/stale" ]] - assert_error_unverified "${root}/m8/paddlefleet_alignment_pin_receipt.json" - - # First two fetches emit Swift 34038640242 curl-56 / early-EOF and leave - # a dest sentinel; the next fetch must see a clean dest. Third fetch is - # real git. Depth=1 and detached pin, not a default-branch clone. - write_git_wrapper "${root}/bin-retry" retry - run PATH="${root}/bin-retry:${root}/bin:${PATH}" \ - ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - bash "${script}" --dest "${root}/ok-retry" - [[ "$(cat "${root}/bin-retry/count")" == 3 ]] - [[ ! -f "${root}/bin-retry/stale" ]] - [[ ! -e "${root}/ok-retry/PaddleFleet/.retry-sentinel" ]] - assert_pin_checkout "${root}/ok-retry/PaddleFleet" "${sha_b}" - python3 - "${root}/ok-retry/paddlefleet_alignment_pin_receipt.json" "${sha_b}" <<'PY' -import json, sys -doc = json.load(open(sys.argv[1])) -assert doc["status"] == "ok" -assert doc["source"]["actual_commit"] == sys.argv[2] -assert doc["source"]["commit_verified"] is True -print("ok-retry receipt exact HEAD verified") -PY - - # Exhausted transient retries: wrapper count is 3, dest cleaned between - # attempts, error receipt stays unverified. - write_git_wrapper "${root}/bin-fail" exhaust - expect_fail "${root}/m9" "git fetch failed" \ - env PATH="${root}/bin-fail:${root}/bin:${PATH}" \ - ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - bash "${script}" --dest "${root}/m9" - [[ "$(cat "${root}/bin-fail/count")" == 3 ]] - [[ ! -f "${root}/bin-fail/stale" ]] - assert_error_unverified "${root}/m9/paddlefleet_alignment_pin_receipt.json" - - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - PADDLEFLEET_WHEEL_URL="${root}/art/py.whl" \ - PADDLEFLEET_WHEEL_SHA256="${py_sha}" \ - PADDLEFLEET_OPS_WHEEL_URL="${root}/art/ops.whl" \ - PADDLEFLEET_OPS_WHEEL_SHA256="${ops_sha}" \ - bash "${script}" --dest "${root}/ok-url" - python3 - "${root}/ok-url/paddlefleet_alignment_pin_receipt.json" "${sha_b}" "${py_sha}" <<'PY' -import json, sys -doc = json.load(open(sys.argv[1])) -sha_b, py_sha = sys.argv[2], sys.argv[3] -assert doc["status"] == "ok" -assert doc["source"]["actual_commit"] == sha_b -assert doc["source"]["commit_verified"] is True -wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") -assert wheel["actual_sha256"] == py_sha -assert wheel["actual_sha256"] != sha_b -assert wheel["digest_verified"] is True -assert wheel.get("built_from_commit") in (None, "") -assert "verified" not in wheel -assert doc["pairing"]["stack_paired_proven"] is False -assert doc["pairing"]["status"] == "unproven" -assert "MinimaxV2.5_EP2" in doc["cases_preserved"] -assert "GLM45Air_EP2" in doc["cases_preserved"] -print("ok-url receipt fields checked") -PY - assert_pin_checkout "${root}/ok-url/PaddleFleet" "${sha_b}" - - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - PADDLEFLEET_BUILD_CMD="mkdir -p '${root}/ok-build/dist' && cp '${root}/art/py.whl' '${root}/ok-build/dist/paddlefleet-0.0.0-py3-none-any.whl'" \ - PADDLEFLEET_BUILD_OPS_CMD="mkdir -p '${root}/ok-build/dist' && cp '${root}/art/ops.whl' '${root}/ok-build/dist/paddlefleet_ops-0.0.0-py3-none-any.whl'" \ - bash "${script}" --dest "${root}/ok-build" - python3 - "${root}/ok-build/paddlefleet_alignment_pin_receipt.json" "${sha_b}" "${py_sha}" <<'PY' -import json, sys -doc = json.load(open(sys.argv[1])) -sha_b, py_sha = sys.argv[2], sys.argv[3] -assert doc["status"] == "ok" -wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") -assert wheel["actual_sha256"] == py_sha -assert wheel["actual_sha256"] != sha_b -assert wheel["origin"] == "build" -assert wheel["built_from_commit"] == sha_b -assert doc["source"]["actual_commit"] == sha_b -assert doc["pairing"]["stack_paired_proven"] is False -assert doc["pairing"]["status"] == "built_from_checked_out_pin" -print("ok-build receipt fields checked") -PY - grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-build/paddlefleet_alignment_pin.env" - - run ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${sha_b}" PADDLEFLEET_GIT_URL="${root}/upstream" \ - bash "${script}" --dest "${root}/ok-source" - python3 - "${root}/ok-source/paddlefleet_alignment_pin_receipt.json" "${sha_b}" "${root}/ok-source" <<'PY' -import json, sys -doc = json.load(open(sys.argv[1])) -sha_b, dest = sys.argv[2], sys.argv[3] -assert doc["status"] == "ok" -assert doc["source"]["actual_commit"] == sha_b -assert doc["source"]["commit_verified"] is True -wheel = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet") -ops = next(a for a in doc["artifacts"] if a["name"] == "paddlefleet_ops") -assert wheel["path"] == f"{dest}/PaddleFleet" -assert ops["path"] == f"{dest}/PaddleFleet/packages/paddlefleet_ops" -assert wheel["origin"] == "source_tree" -assert ops["origin"] == "source_tree" -assert doc["pairing"]["status"] == "source_tree_from_checked_out_pin" -assert doc["pairing"]["stack_paired_proven"] is False -print("ok-source receipt fields checked") -PY - grep -q "PADDLEFLEET_WHEEL_PATH=${root}/ok-source/PaddleFleet$" "${root}/ok-source/paddlefleet_alignment_pin.env" - grep -q "PADDLEFLEET_OPS_WHEEL_PATH=${root}/ok-source/PaddleFleet/packages/paddlefleet_ops$" "${root}/ok-source/paddlefleet_alignment_pin.env" - grep -q "PADDLEFLEET_WHEEL_ORIGIN=source_tree" "${root}/ok-source/paddlefleet_alignment_pin.env" - grep -q "PADDLEFLEET_WHEEL_DIGEST_VERIFIED=false" "${root}/ok-source/paddlefleet_alignment_pin.env" - grep -q "PADDLEFLEET_SOURCE_COMMIT=${sha_b}" "${root}/ok-source/paddlefleet_alignment_pin.env" - assert_pin_checkout "${root}/ok-source/PaddleFleet" "${sha_b}" - - grep -q 'CodeSync/develop/PaddleFleet.tar' "${script}" - grep -q 'PaddleFleet/develop/latest/paddlefleet-0.0.0-py3-none-linux_x86_64.whl' "${script}" - - echo "select_paddlefleet_alignment_pin self-test OK" -} - -if [[ "${RUN_SELF_TEST}" == 1 ]]; then - run_self_test - exit 0 -fi - -mkdir -p "${DEST}" -case "${MODE}" in - stack-paired) fetch_stack_paired ;; - develop) fetch_default ;; - *) fail "unknown ALIGNMENT_PADDLEFLEET_MODE=${MODE} (develop|stack-paired)" ;; -esac diff --git a/scripts/test_alignment_workflow_shell.sh b/scripts/test_alignment_workflow_shell.sh deleted file mode 100755 index 196e32c4ad..0000000000 --- a/scripts/test_alignment_workflow_shell.sh +++ /dev/null @@ -1,161 +0,0 @@ -#!/usr/bin/env bash -# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. -# -# Syntax-check every workflow `run:` block, then execute the extracted -# Get Whl docker-exec body against a failing selector. Independent -# helper --self-test is not this check. - -set -euo pipefail - -ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" -YAML="${ROOT}/../.github/workflows/alignment_model_accuracy.yaml" -SELECTOR_PATH="/workspace/ms-swift/scripts/select_paddlefleet_alignment_pin.sh" -REQUIRE_PATH="/workspace/ms-swift/scripts/require_paddlefleet_selector_ok.sh" -BUILD_MARKER="build ms-swift" - -python3 - "${YAML}" "${ROOT}" "${SELECTOR_PATH}" "${REQUIRE_PATH}" "${BUILD_MARKER}" <<'PY' -import os, re, subprocess, sys, tempfile, textwrap, pathlib, json, stat - -yaml_path, scripts_root, selector_path, require_path, build_marker = sys.argv[1:6] -text = pathlib.Path(yaml_path).read_text() -lines = text.splitlines(True) - -# Extract literal `run: |` blocks by indent. -blocks = [] -i = 0 -while i < len(lines): - m = re.match(r"^(\s*)run:\s*\|\s*$", lines[i]) - if not m: - i += 1 - continue - indent = len(m.group(1)) - name = "unnamed" - for j in range(i, -1, -1): - nm = re.match(r"^\s+- name:\s*(.*)$", lines[j]) - if nm: - name = nm.group(1).strip() - break - i += 1 - body = [] - while i < len(lines): - line = lines[i] - if line.strip() == "": - body.append(line) - i += 1 - continue - lead = len(line) - len(line.lstrip(" ")) - if lead <= indent and line.strip(): - break - body.append(line[indent + 2 :] if lead >= indent + 2 else line.lstrip()) - i += 1 - blocks.append((name, "".join(body))) - -if not blocks: - raise SystemExit(f"no run: | blocks in {yaml_path}") - -tmp = pathlib.Path(tempfile.mkdtemp(prefix="yaml-run-")) -print(f"extracted {len(blocks)} run blocks from {yaml_path}") -for n, body in blocks: - p = tmp / (re.sub(r"[^A-Za-z0-9._-]+", "_", n) + ".sh") - p.write_text("#!/usr/bin/env bash\n" + body) - r = subprocess.run(["bash", "-n", str(p)], capture_output=True, text=True) - if r.returncode != 0: - raise SystemExit(f"bash -n FAIL {n}: {r.stderr}") - print(f"bash -n OK run:{n}") - - # Reconstruct docker exec -c quoting: the host sees - # docker exec ... /bin/bash -c 'INNER' - # INNER must not contain an unescaped single quote. - for m in re.finditer(r"""/bin/bash -c\s+'""", body): - start = m.end() - end = body.find("'\n", start) - if end < 0: - end = body.rfind("'") - inner = body[start:end] - if "'" in inner: - raise SystemExit( - f"nested single quote inside docker exec -c in {n}: " - f"{inner[inner.find(chr(39))-40:inner.find(chr(39))+40]!r}" - ) - inner_p = tmp / (p.stem + ".docker-inner.sh") - inner_p.write_text("#!/usr/bin/env bash\n" + inner + "\n") - r = subprocess.run(["bash", "-n", str(inner_p)], capture_output=True, text=True) - if r.returncode != 0: - raise SystemExit(f"bash -n FAIL docker-inner {n}: {r.stderr}") - print(f"bash -n OK docker-inner:{n} (no nested single quotes)") - -getwhl = next((b for n, b in blocks if n == "Get Whl"), None) -if getwhl is None: - raise SystemExit("Get Whl run block missing") -m = re.search(r"""/bin/bash -c\s+'""", getwhl) -if not m: - raise SystemExit("Get Whl docker exec -c missing") -inner = getwhl[m.end():] -end = inner.rfind("'") -inner = inner[:end] - -# Fixture: run the extracted Get Whl body with a failing selector. -ws = tmp / "ws" -(ws / "ms-swift/scripts").mkdir(parents=True) -(ws / "upload").mkdir(parents=True) -selector = ws / "ms-swift/scripts/select_paddlefleet_alignment_pin.sh" -require_src = pathlib.Path(scripts_root) / "require_paddlefleet_selector_ok.sh" -require_dst = ws / "ms-swift/scripts/require_paddlefleet_selector_ok.sh" -require_dst.write_text(require_src.read_text()) -require_dst.chmod(require_dst.stat().st_mode | stat.S_IXUSR) -selector.write_text(textwrap.dedent("""\ - #!/usr/bin/env bash - set -euo pipefail - dest="${2:-/workspace}" - mkdir -p "${dest}" - cat >"${dest}/paddlefleet_alignment_pin_receipt.json" <<'EOF' - {"schema":"paddlefleet-alignment-pin/v1","status":"error","detail":"git clone failed: github.com:443","mode":"stack-paired"} - EOF - echo "[paddlefleet-pin] FAIL: git clone failed: github.com:443" >&2 - echo "::error:: git clone failed: github.com:443" >&2 - exit 1 - """)) -selector.chmod(selector.stat().st_mode | stat.S_IXUSR) -(ws / "ms-swift/scripts/dependence").mkdir(parents=True, exist_ok=True) -(ws / "ms-swift/scripts/dependence/build.sh").write_text("#!/usr/bin/env bash\necho BUILD_RAN > /workspace/upload/BUILD_RAN\n") -(ws / "ms-swift/scripts/dependence/build.sh").chmod(0o755) - -# Rewrite extracted inner to use the fixture workspace and stub tools. -rewritten = inner -rewritten = rewritten.replace("/workspace", str(ws)) -rewritten = rewritten.replace("conda activate py_$python_version", "true") -rewritten = rewritten.replace(". /opt/conda/etc/profile.d/conda.sh", "true") -rewritten = "#!/usr/bin/env bash\nexport ALIGNMENT_PADDLEFLEET_MODE=stack-paired\nexport python_version=3.12\n" + rewritten -# Stub wget / pip / python / ldconfig so a leaked continue would be visible. -bin = tmp / "bin" -bin.mkdir() -(bin / "wget").write_text("#!/usr/bin/env bash\necho WGET_RAN \"$@\" >> '%s/WGET_RAN'\nexit 0\n" % ws) -(bin / "python").write_text("#!/usr/bin/env bash\necho PY_RAN \"$@\" >> '%s/PY_RAN'\nexit 0\n" % ws) -(bin / "python3").write_text("#!/usr/bin/env bash\nexec /usr/bin/python3 \"$@\"\n") -(bin / "pip").write_text("#!/usr/bin/env bash\necho PIP_RAN \"$@\" >> '%s/PIP_RAN'\nexit 0\n" % ws) -(bin / "ldconfig").write_text("#!/usr/bin/env bash\nexit 0\n") -for f in bin.iterdir(): - f.chmod(0o755) - -script = tmp / "getwhl.extracted.sh" -script.write_text(rewritten) -script.chmod(0o755) -env = os.environ.copy() -env["PATH"] = str(bin) + ":" + env.get("PATH", "") -env["ALIGNMENT_PADDLEFLEET_MODE"] = "stack-paired" -r = subprocess.run(["bash", str(script)], capture_output=True, text=True, env=env) -log = (r.stdout or "") + (r.stderr or "") -print("extracted Get Whl rc=", r.returncode) -print(log[-2000:]) -if r.returncode == 0: - raise SystemExit("FAIL: extracted Get Whl continued after selector failure") -if (ws / "WGET_RAN").exists(): - raise SystemExit("FAIL: wget ran after selector failure") -if (ws / "upload/BUILD_RAN").exists(): - raise SystemExit("FAIL: build.sh ran after selector failure") -if "selector failed; stop Get Whl" not in log and "git clone failed" not in log: - raise SystemExit("FAIL: extracted Get Whl did not surface selector failure") -print("extracted Get Whl fixture: selector fail stopped remaining wheels and build") -print("workflow shell checks OK") -PY -echo "alignment workflow shell PATH_PASS (extracted YAML, not helper-only)" diff --git a/scripts/test_checkout_no_pager.sh b/scripts/test_checkout_no_pager.sh deleted file mode 100755 index 21a1335b65..0000000000 --- a/scripts/test_checkout_no_pager.sh +++ /dev/null @@ -1,65 +0,0 @@ -#!/usr/bin/env bash -# Reproduce an interactive Git pager in a real PTY, then check the workflow command. -set -euo pipefail -ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" -"${PYTHON_BIN:-python3}" - "$ROOT" <<'PYTEST' -import fcntl -import os -import pathlib -import pty -import select -import shlex -import shutil -import signal -import struct -import subprocess -import sys -import termios -import time - -root = pathlib.Path(sys.argv[1]) -text = (root / ".github/workflows/alignment_model_accuracy.yaml").read_text() -line = next(line.strip() for line in text.splitlines() if "checkout_step swift-history " in line) -actual = shlex.split(line)[3:] -assert actual == ["git", "--no-pager", "log", "--pretty=oneline", "-10"], actual -assert shutil.which("less"), "PTY regression fixture requires less" - -def observe(command): - master, slave = pty.openpty() - fcntl.ioctl(slave, termios.TIOCSWINSZ, struct.pack("HHHH", 8, 60, 0, 0)) - env = dict(os.environ, TERM="xterm", GIT_PAGER="less", LESS="-FRX") - proc = subprocess.Popen(command, cwd=root, env=env, stdin=slave, stdout=slave, - stderr=slave, start_new_session=True) - os.close(slave) - output = bytearray() - try: - until = time.monotonic() + 2 - while time.monotonic() < until and proc.poll() is None: - if select.select([master], [], [], 0.05)[0]: - try: - output.extend(os.read(master, 65536)) - except OSError: - break - # PTY EOF can arrive just before the child is reaped. - try: - proc.wait(timeout=0.2) - needed_input = False - except subprocess.TimeoutExpired: - needed_input = True - if needed_input: - os.write(master, b"q") - result = proc.wait(timeout=3) - assert result == 0, (result, bytes(output)[-200:]) - assert output, "expected Git history output" - return needed_input - finally: - if proc.poll() is None: - os.killpg(proc.pid, signal.SIGKILL) - proc.wait() - os.close(master) - -assert observe(["git", "log", "--pretty=oneline", "-10"]), "old command should await pager input in PTY" -print("PASS: old git log waits for input in PTY and exits on q") -assert not observe(actual), "workflow Git history must exit without input" -print("PASS: actual workflow --no-pager command exits without input in same PTY") -PYTEST diff --git a/scripts/test_checkout_observability.sh b/scripts/test_checkout_observability.sh deleted file mode 100755 index 6d954a273f..0000000000 --- a/scripts/test_checkout_observability.sh +++ /dev/null @@ -1,60 +0,0 @@ -#!/usr/bin/env bash -# Exercise the checkout wrapper extracted from the actual workflow, without network access. -set -euo pipefail -ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" -PYTHON_BIN="${PYTHON_BIN:-python3}" -"$PYTHON_BIN" - "$ROOT/.github/workflows/alignment_model_accuracy.yaml" <<'PYTEST' -import pathlib -import re -import shlex -import subprocess -import sys -import tempfile -import textwrap -import time - -text = pathlib.Path(sys.argv[1]).read_text() -checkout = text.split(" - name: Checkout Code\n", 1)[1].split(" - name:", 1)[0] -match = re.search(r"^ checkout_step\(\) \{\n.*?^ \}\n", checkout, re.M | re.S) -assert match, "actual checkout wrapper missing" -helper = textwrap.dedent(match.group()) -subprocess.run(["bash", "-n"], input=helper, text=True, check=True) -assert "set -eo pipefail" in checkout -assert not re.search(r"^\s*(?:wget |git (?:pull|fetch|submodule) )", checkout, re.M), "unbounded network command" -assert 'test "$(git rev-parse HEAD)" = "$COMMIT_ID"' in checkout, "exact HEAD check removed" - -secret = "fixture-credential-must-not-be-logged" -def run(label, limit, command): - invocation = shlex.join(["checkout_step", label, limit, *command, secret]) - script = helper + "\nset -e\n" + invocation + "\necho reached-next-step\n" - started = time.monotonic() - result = subprocess.run(["bash"], input=script, text=True, capture_output=True, timeout=8) - elapsed = time.monotonic() - started - output = result.stdout + result.stderr - assert secret not in output, output - assert "checkout begin: " + label in output, output - assert "checkout end: " + label in output, output - return result.returncode, output, elapsed - -rc, output, _ = run("success", "2s", ["bash", "-c", "exit 0"]) -assert rc == 0 and "status=0" in output and "reached-next-step" in output, output -print("PASS: success logs begin/end and continues without printing arguments") - -rc, output, _ = run("failure", "2s", ["bash", "-c", "exit 7"]) -assert rc == 7 and "checkout failed: failure status=7" in output and "reached-next-step" not in output, output -print("PASS: command failure preserves exit status and stops checkout") - -with tempfile.TemporaryDirectory(prefix="checkout-timeout-") as directory: - marker = pathlib.Path(directory) / "marker" - command = 'printf started > "$1"; sleep 20; printf finished > "$1"' - rc, output, elapsed = run("timeout", "0.2s", ["bash", "-c", command, "fixture", str(marker)]) - assert marker.read_text() == "started", output - assert rc == 124 and elapsed < 5, (rc, elapsed, output) - assert "checkout timed out: timeout" in output and "reached-next-step" not in output, output -print("PASS: actual timeout ends a running command and prevents continuation") - -rc, output, _ = run("killed", "2s", ["bash", "-c", "exit 137"]) -assert rc == 137 and "checkout terminated: killed" in output and "checkout timed out:" not in output, output -print("PASS: exit 137 is not falsely identified as a proven timeout") -print("All checkout observability fixtures passed") -PYTEST diff --git a/scripts/test_paddlefleet_pin_handoff.sh b/scripts/test_paddlefleet_pin_handoff.sh deleted file mode 100755 index 72cc719998..0000000000 --- a/scripts/test_paddlefleet_pin_handoff.sh +++ /dev/null @@ -1,114 +0,0 @@ -#!/usr/bin/env bash -# Copyright (c) 2026 PaddlePaddle Authors. All Rights Reserved. -# -# End-to-end path check: selector source-mode -> new-shell consume -> -# setup_venvs *path* consumer. This proves docker-exec handoff of source -# paths. It is NOT a uv install / real setup_venvs / numerical CI run. -# Isolated selector --self-test is not enough for the step boundary. - -set -euo pipefail - -ROOT="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" && pwd)" -SELECTOR="${ROOT}/select_paddlefleet_alignment_pin.sh" -CONSUME="${ROOT}/consume_paddlefleet_alignment_pin.sh" - -tmp="$(mktemp -d)" -trap 'rm -rf "${tmp}"' EXIT - -git init -q "${tmp}/upstream" -git -C "${tmp}/upstream" config user.email test@example.com -git -C "${tmp}/upstream" config user.name test -mkdir -p "${tmp}/upstream/packages/paddlefleet_ops" -printf '%s\n' '[project]' 'name = "paddlefleet"' >"${tmp}/upstream/pyproject.toml" -printf '%s\n' '[project]' 'name = "paddlefleet-ops"' >"${tmp}/upstream/packages/paddlefleet_ops/pyproject.toml" -echo src >"${tmp}/upstream/README" -git -C "${tmp}/upstream" add README pyproject.toml packages -git -C "${tmp}/upstream" commit -q -m pin -PIN="$(git -C "${tmp}/upstream" rev-parse HEAD)" - -# Step A: Get Whl equivalent (selector). -ALIGNMENT_PADDLEFLEET_MODE=stack-paired \ - PADDLEFLEET_PIN_SHA="${PIN}" \ - PADDLEFLEET_GIT_URL="${tmp}/upstream" \ - bash "${SELECTOR}" --dest "${tmp}/ws" - -test -f "${tmp}/ws/paddlefleet_alignment_pin.env" -test -d "${tmp}/ws/PaddleFleet" - -# Step B: new docker exec — drop selector shell state, keep only files. -# Requested mode/pin stay on the caller; leftover develop env must not win. -unset PADDLEFLEET_WHEEL_PATH PADDLEFLEET_OPS_WHEEL_PATH PADDLEFLEET_SOURCE_COMMIT || true -ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${PIN}" \ - bash "${CONSUME}" --env "${tmp}/ws/paddlefleet_alignment_pin.env" --out "${tmp}/ws/consumed.env" - -# Negative: leftover develop env + requested stack-paired must fail closed. -mkdir -p "${tmp}/dev" -echo dummy >"${tmp}/dev/paddlefleet-0.0.0-py3-none-linux_x86_64.whl" -echo dummy >"${tmp}/dev/paddlefleet_ops-0.0.0-cp312-cp312-linux_x86_64.whl" -cat >"${tmp}/dev.env" <&2 - exit 1 -fi - -# Negative: selector clone/fail writes error receipt and no env. Consume -# must name selector failure. Missing env here is the consequence, not -# proof that a generated env failed to cross docker exec. -REQUIRE="${ROOT}/require_paddlefleet_selector_ok.sh" -mkdir -p "${tmp}/sel-fail" -cat >"${tmp}/sel-fail/paddlefleet_alignment_pin_receipt.json" <<'EOF' -{"schema":"paddlefleet-alignment-pin/v1","status":"error","detail":"git clone failed: github.com:443","mode":"stack-paired"} -EOF -if bash "${REQUIRE}" "${tmp}/sel-fail" >"${tmp}/sel-fail.require.out" 2>"${tmp}/sel-fail.require.err"; then - echo "handoff FAIL: require_ok accepted error receipt" >&2 - exit 1 -fi -grep -q "status='error' is not ok" "${tmp}/sel-fail.require.err" \ - || grep -q 'status="error" is not ok' "${tmp}/sel-fail.require.err" \ - || grep -q "status=error is not ok" "${tmp}/sel-fail.require.err" \ - || grep -q "selector receipt status=" "${tmp}/sel-fail.require.err" -if ALIGNMENT_PADDLEFLEET_MODE=stack-paired PADDLEFLEET_PIN_SHA="${PIN}" \ - bash "${CONSUME}" --env "${tmp}/sel-fail/paddlefleet_alignment_pin.env" \ - --out "${tmp}/sel-fail.consumed.env" 2>"${tmp}/sel-fail.err"; then - echo "handoff FAIL: selector error receipt was consumed" >&2 - exit 1 -fi -grep -q "because selector failed" "${tmp}/sel-fail.err" -if grep -q "selector wrote ok receipt but env did not cross docker exec" "${tmp}/sel-fail.err"; then - echo "handoff FAIL: selector error misclassified as env-handoff" >&2 - exit 1 -fi - -# Step C: path consumer only. Does not run uv or setup_venvs.sh. -stub_setup="${tmp}/setup_path_consumer.sh" -cat >"${stub_setup}" <<'STUB' -#!/usr/bin/env bash -set -euo pipefail -# Mirrors setup_venvs.sh reading PADDLEFLEET_WHEEL_PATH. Path presence only. -PADDLEFLEET_WHEEL="${PADDLEFLEET_WHEEL_PATH:?missing PADDLEFLEET_WHEEL_PATH}" -PADDLEFLEET_OPS_WHEEL="${PADDLEFLEET_OPS_WHEEL_PATH:?missing PADDLEFLEET_OPS_WHEEL_PATH}" -[[ -d "${PADDLEFLEET_WHEEL}" || -f "${PADDLEFLEET_WHEEL}" ]] || { echo "missing ${PADDLEFLEET_WHEEL}" >&2; exit 1; } -[[ -d "${PADDLEFLEET_OPS_WHEEL}" || -f "${PADDLEFLEET_OPS_WHEEL}" ]] || { echo "missing ${PADDLEFLEET_OPS_WHEEL}" >&2; exit 1; } -echo "PATH_CONSUMER paddlefleet=${PADDLEFLEET_WHEEL}" -echo "PATH_CONSUMER ops=${PADDLEFLEET_OPS_WHEEL}" -echo "PATH_CONSUMER not_uv_install=true" -STUB -chmod +x "${stub_setup}" - -set -a -# shellcheck disable=SC1090 -. "${tmp}/ws/consumed.env" -set +a -bash "${stub_setup}" | tee "${tmp}/setup.out" -grep -q "PATH_CONSUMER paddlefleet=${tmp}/ws/PaddleFleet" "${tmp}/setup.out" -grep -q "packages/paddlefleet_ops" "${tmp}/setup.out" -grep -q "not_uv_install=true" "${tmp}/setup.out" - -echo "paddlefleet pin handoff PATH_PASS pin=${PIN} (not uv install, not CI)" From 91a3bb86d72fb8e301bb1a920b7815846c02820e Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Wed, 9 Sep 2026 14:01:47 +0800 Subject: [PATCH 27/33] test: verify YAML alignment mode reaches Megatron config Signed-off-by: Zhan Rongrui --- tests/megatron/test_model_config.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 750745c986..a2d98337e3 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -47,6 +47,20 @@ def test_save_processor_prefers_independent_tokenizer_source(): assert _get_save_processor_id(SimpleNamespace(model_dir='/weights', tokenizer_name_or_path=None)) == '/weights' +def test_get_mcore_model_config_propagates_accuracy_mode(monkeypatch): + # Preserve the real inherited dataclass fields while avoiding model construction. + actual_fields = utils.fields(utils.ModelConfig) + assert 'use_accuracy_compatible' in {field.name for field in actual_fields} + monkeypatch.setattr(utils, 'ModelConfig', _ModelConfigStub) + monkeypatch.setattr(utils, 'fields', lambda _: actual_fields) + for enabled in (False, True, False): + args = _make_args() + args.use_accuracy_compatible = enabled + monkeypatch.setenv('USE_ACCURACY_COMPATIBLE', str(int(not enabled))) + config = utils.get_mcore_model_config(args, PretrainedConfig()) + assert config.kwargs['use_accuracy_compatible'] is enabled + + def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) From 47d17bfbc26e9f6ab439a9e31835a46e8fb91b63 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 10 Sep 2026 10:37:33 +0800 Subject: [PATCH 28/33] style: fix alignment workflow whitespace and script ending --- .github/workflows/alignment_model_accuracy.yaml | 4 ++-- scripts/dependence/build.sh | 1 - 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index 76faf444ba..6de447d2e4 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -65,7 +65,7 @@ jobs: wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar cd PaddleFleet && git pull && cd - - + echo "Download ms-swift form https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz" wget -q --no-proxy https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate rm -rf ms-swift && tar zxf ms-swift.tar.gz && rm -rf ms-swift.tar.gz @@ -145,7 +145,7 @@ jobs: export MEGATRON_CORE_WHEEL_PATH=/workspace/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl export MS_SWIFT_WHEEL_PATH=/workspace/upload/ms_swift-0.0.0-py3-none-any.whl export MCORE_BRIDGE_WHEEL_PATH=/workspace/mcore_bridge-0.0.0-py3-none-any.whl - # The BOS wheels are republished under a fixed 0.0.0 filename + # The BOS wheels are republished under a fixed 0.0.0 filename export UV_SKIP_WHEEL_FILENAME_CHECK=1 for whl in "$PADDLEFLEET_WHEEL_PATH" "$PADDLEFLEET_OPS_WHEEL_PATH" \ diff --git a/scripts/dependence/build.sh b/scripts/dependence/build.sh index de73702692..6b9129513c 100644 --- a/scripts/dependence/build.sh +++ b/scripts/dependence/build.sh @@ -79,4 +79,3 @@ echo -e "\033[32m ---- make ms-swift.tar.gz \033[0m" swift_tar echo -e "\033[32m ---- build ms-swift whl \033[0m" swift_build - From cd14a9e67cf68cf9bc861c0f24a44ffd0aab0ec6 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 10 Sep 2026 16:35:06 +0800 Subject: [PATCH 29/33] fix(megatron): preserve model-specific accuracy loss accumulation --- swift/megatron/model/utils.py | 4 +++ swift/megatron/trainers/trainer.py | 7 +++- tests/megatron/test_accuracy_loss_and_norm.py | 33 +++++++++++++++---- tests/megatron/test_model_config.py | 21 ++++++++++++ 4 files changed, 57 insertions(+), 8 deletions(-) diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index ca387b0f12..b813bb7646 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -56,6 +56,8 @@ def get_mcore_model_config(args, hf_config): kwargs = hf_to_mcore_config(hf_config) llm_config = HfConfigFactory.get_text_config(hf_config) n_routed_experts = getattr(llm_config, 'n_routed_experts', None) + if getattr(llm_config, 'model_type', None) == 'glm_moe_dsa': + kwargs['accuracy_compatible_loss_sum_dtype'] = 'float32' if n_routed_experts is not None: kwargs['num_moe_experts'] = n_routed_experts if getattr(args, 'mtp_num_layers', None) is None: @@ -97,6 +99,8 @@ def get_mcore_model_config(args, hf_config): kwargs['moe_enable_routing_replay'] = True if args.megatron_extra_kwargs: kwargs.update(args.megatron_extra_kwargs) + if kwargs.get('accuracy_compatible_loss_sum_dtype', 'float64') not in {'float32', 'float64'}: + raise ValueError('accuracy_compatible_loss_sum_dtype must be float32 or float64') config = ModelConfig(**kwargs) if is_torch_npu_available() and getattr(args, 'attention_backend', 'flash') != 'local': setattr(config, 'use_flash_attn', True) diff --git a/swift/megatron/trainers/trainer.py b/swift/megatron/trainers/trainer.py index 62a77c69d5..035bbd36a4 100644 --- a/swift/megatron/trainers/trainer.py +++ b/swift/megatron/trainers/trainer.py @@ -385,7 +385,12 @@ def loss_func(self, losses = losses * torch.exp(-losses.detach()) if loss_scale is not None: losses = losses * loss_scale - loss = torch.cat([torch.sum(losses * loss_mask).view(1), loss_mask.sum().view(1)]) + masked_losses = losses * loss_mask + if _use_accuracy_compatible_enabled() and self.config.accuracy_compatible_loss_sum_dtype == 'float64': + loss_sum = masked_losses.reshape(-1).double().sum().float() + else: + loss_sum = torch.sum(masked_losses) + loss = torch.cat([loss_sum.view(1), loss_mask.sum().view(1)]) # Reduce loss for logging. reporting_loss = loss.detach().clone() diff --git a/tests/megatron/test_accuracy_loss_and_norm.py b/tests/megatron/test_accuracy_loss_and_norm.py index 506c00de52..57241f7900 100644 --- a/tests/megatron/test_accuracy_loss_and_norm.py +++ b/tests/megatron/test_accuracy_loss_and_norm.py @@ -10,6 +10,8 @@ from pathlib import Path from unittest.mock import patch +from swift.megatron.trainers import trainer as trainer_module + ROOT = Path(__file__).resolve().parents[2] @@ -34,20 +36,18 @@ def test_mask_scale_local_gradient_and_global_reporting(self): torch.cuda.set_device(0) for enabled in (False, True): with self.subTest(accuracy=enabled): - loss_func = production_function( - 'swift/megatron/trainers/trainer.py', 'loss_func', { - 'torch': torch, - 'mpu': types.SimpleNamespace(get_data_parallel_group=lambda **kwargs: None), - '_use_accuracy_compatible_enabled': lambda: enabled, - }) + loss_func = trainer_module.MegatronTrainer.loss_func trainer = types.SimpleNamespace( - args=types.SimpleNamespace(enable_dft_loss=False, enable_channel_loss=False)) + args=types.SimpleNamespace(enable_dft_loss=False, enable_channel_loss=False), + config=types.SimpleNamespace(accuracy_compatible_loss_sum_dtype='float64')) values = torch.tensor([[2., 19., 3.]], device='cuda', requires_grad=True) labels = torch.tensor([[1, -100, 2]], device='cuda') scale = torch.tensor([[0.5, 1000., 2.]], device='cuda') # A second identical DP rank contributes to reporting, not local backward. with patch.object(torch.distributed, 'all_reduce', side_effect=lambda value, **kwargs: value.mul_(2)), \ patch.object(torch.distributed, 'get_rank', return_value=0), \ + patch.object(trainer_module.mpu, 'get_data_parallel_group', return_value=None), \ + patch.object(trainer_module, '_use_accuracy_compatible_enabled', return_value=enabled), \ contextlib.redirect_stdout(io.StringIO()): loss, count, metrics = loss_func(trainer, values, labels=labels, loss_scale=scale) self.assertEqual(loss.item(), 7.) @@ -57,6 +57,25 @@ def test_mask_scale_local_gradient_and_global_reporting(self): loss.backward() self.assertEqual(values.grad.tolist(), [[0.5, 0., 2.]]) + def test_fp64_compatibility_sum_keeps_small_losses_and_masked_gradients(self): + torch.cuda.set_device(0) + trainer = types.SimpleNamespace( + args=types.SimpleNamespace(enable_dft_loss=False, enable_channel_loss=False), + config=types.SimpleNamespace(accuracy_compatible_loss_sum_dtype='float64')) + values = torch.tensor([[100000000.] + [1.] * 8 + [100000000.]], device='cuda', requires_grad=True) + labels = torch.tensor([[1] * 9 + [-100]], device='cuda') + with patch.object(torch.distributed, 'all_reduce'), \ + patch.object(torch.distributed, 'get_rank', return_value=0), \ + patch.object(trainer_module.mpu, 'get_data_parallel_group', return_value=None), \ + patch.object(trainer_module, '_use_accuracy_compatible_enabled', return_value=True), \ + contextlib.redirect_stdout(io.StringIO()): + loss, count, metrics = trainer_module.MegatronTrainer.loss_func(trainer, values, labels=labels) + self.assertEqual(loss.item(), 100000008.) + self.assertEqual(count.item(), 9) + self.assertEqual(metrics['loss'].tolist(), [100000008., 9.]) + loss.backward() + self.assertEqual(values.grad.tolist(), [[1.] * 9 + [0.]]) + def test_indexer_norm_preserves_disabled_provider(self): native_norm = type('NativeNorm', (), {}) provider_norm = type('ProviderNorm', (), {}) diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index a2d98337e3..e96b8046b1 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -70,6 +70,27 @@ def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): assert config.kwargs['mtp_num_layers'] == 1 +def test_glm52_loss_sum_contract_keeps_other_models_default(monkeypatch): + _patch_model_config(monkeypatch) + for model_type in ('glm_moe_dsa', 'glm4_moe', 'minimax_m2'): + hf_config = PretrainedConfig(model_type=model_type) + config = utils.get_mcore_model_config(_make_args(), hf_config) + assert config.kwargs.get('accuracy_compatible_loss_sum_dtype', + 'float64') == ('float32' if model_type == 'glm_moe_dsa' else 'float64') + + +def test_loss_sum_contract_rejects_unsupported_dtype(monkeypatch): + _patch_model_config(monkeypatch) + args = _make_args() + args.megatron_extra_kwargs = {'accuracy_compatible_loss_sum_dtype': 'bfloat16'} + try: + utils.get_mcore_model_config(args, PretrainedConfig()) + except ValueError as error: + assert 'accuracy_compatible_loss_sum_dtype' in str(error) + else: + raise AssertionError('BF16 loss accumulation must fail before constructing the model') + + def test_get_mcore_model_config_reads_mtp_num_hidden_layers(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(text_config=PretrainedConfig(mtp_num_hidden_layers=1)) From 4dc94a9e616bc16d5dd5e9767925b4e6e204cbc2 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 10 Sep 2026 16:45:34 +0800 Subject: [PATCH 30/33] fix(megatron): keep MTP training explicitly configured --- swift/megatron/model/utils.py | 14 ++------------ tests/megatron/test_model_config.py | 14 +++++++------- 2 files changed, 9 insertions(+), 19 deletions(-) diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index b813bb7646..e9ba83267a 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -44,14 +44,6 @@ def _check_dsa_index_share_recompute(config): 'replayed without its source computing layer. Set recompute_granularity=none.') -def _get_hf_mtp_num_layers(hf_config): - llm_config = HfConfigFactory.get_text_config(hf_config) - for key in ['num_nextn_predict_layers', 'mtp_num_hidden_layers']: - value = getattr(llm_config, key, None) - if value is not None: - return value - - def get_mcore_model_config(args, hf_config): kwargs = hf_to_mcore_config(hf_config) llm_config = HfConfigFactory.get_text_config(hf_config) @@ -60,10 +52,8 @@ def get_mcore_model_config(args, hf_config): kwargs['accuracy_compatible_loss_sum_dtype'] = 'float32' if n_routed_experts is not None: kwargs['num_moe_experts'] = n_routed_experts - if getattr(args, 'mtp_num_layers', None) is None: - mtp_num_layers = _get_hf_mtp_num_layers(hf_config) - if mtp_num_layers is not None: - kwargs['mtp_num_layers'] = mtp_num_layers + # Checkpoint MTP metadata describes available weights, not an opt-in to + # auxiliary training. The explicit mtp_num_layers argument below controls it. kwargs['mcore_model_type'] = args.megatron_model_meta.model_type kwargs['hf_config'] = hf_config for f in fields(ModelConfig): diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index e96b8046b1..61f0d5bfb8 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -61,13 +61,13 @@ def test_get_mcore_model_config_propagates_accuracy_mode(monkeypatch): assert config.kwargs['use_accuracy_compatible'] is enabled -def test_get_mcore_model_config_reads_mtp_num_layers_from_hf(monkeypatch): +def test_get_mcore_model_config_does_not_enable_mtp_from_checkpoint(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) config = utils.get_mcore_model_config(_make_args(), hf_config) - assert config.kwargs['mtp_num_layers'] == 1 + assert not config.kwargs.get('mtp_num_layers') def test_glm52_loss_sum_contract_keeps_other_models_default(monkeypatch): @@ -91,13 +91,13 @@ def test_loss_sum_contract_rejects_unsupported_dtype(monkeypatch): raise AssertionError('BF16 loss accumulation must fail before constructing the model') -def test_get_mcore_model_config_reads_mtp_num_hidden_layers(monkeypatch): +def test_get_mcore_model_config_does_not_enable_mtp_from_nested_checkpoint(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(text_config=PretrainedConfig(mtp_num_hidden_layers=1)) config = utils.get_mcore_model_config(_make_args(), hf_config) - assert config.kwargs['mtp_num_layers'] == 1 + assert not config.kwargs.get('mtp_num_layers') def test_get_mcore_model_config_prefers_n_routed_experts(monkeypatch): @@ -113,9 +113,9 @@ def test_get_mcore_model_config_keeps_explicit_mtp_num_layers(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) - config = utils.get_mcore_model_config(_make_args(mtp_num_layers=2), hf_config) - - assert config.kwargs['mtp_num_layers'] == 2 + for depth in (0, 1, 2): + config = utils.get_mcore_model_config(_make_args(mtp_num_layers=depth), hf_config) + assert config.kwargs['mtp_num_layers'] == depth def test_get_padding_to_sequence_parallel_uses_tp_times_two(): From 2bea97f4e001035d9ed69ad72d12fd6217248f89 Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Thu, 10 Sep 2026 18:14:03 +0800 Subject: [PATCH 31/33] fix: preserve configured gradient clipping in accuracy mode --- swift/megatron/pipelines/train/sft.py | 3 --- tests/megatron/test_model_config.py | 31 +++++++++++++++++++++++++++ 2 files changed, 31 insertions(+), 3 deletions(-) diff --git a/swift/megatron/pipelines/train/sft.py b/swift/megatron/pipelines/train/sft.py index 381a02b24d..9b007a2f14 100644 --- a/swift/megatron/pipelines/train/sft.py +++ b/swift/megatron/pipelines/train/sft.py @@ -43,9 +43,6 @@ def __init__(self, args: Optional[Union[List[str], MegatronSftArguments]] = None self.train_msg = {} super(SwiftSft, self).__init__(args) args = self.args - from megatron.core.transformer.module import _use_accuracy_compatible - if _use_accuracy_compatible() and getattr(args, 'clip_grad', 0) > 0: - args.clip_grad = 0.0 if repatch is not None: megatron_args = asdict(self.args) if args.attention_backend != 'local': diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 61f0d5bfb8..77f6f5315c 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -61,6 +61,37 @@ def test_get_mcore_model_config_propagates_accuracy_mode(monkeypatch): assert config.kwargs['use_accuracy_compatible'] is enabled +def test_sft_pipeline_preserves_explicit_gradient_clipping(monkeypatch): + from swift.megatron.pipelines.train import sft + + def initialize_arguments(pipeline, args): + pipeline.args = args + + def prepare_template(pipeline): + pipeline.template = SimpleNamespace() + + monkeypatch.setattr(sft.SwiftSft.__mro__[1], '__init__', initialize_arguments) + monkeypatch.setattr(sft.MegatronSft, '_prepare_template', prepare_template) + monkeypatch.setattr(sft, 'repatch', None) + for accuracy_enabled in (False, True): + monkeypatch.setenv('USE_ACCURACY_COMPATIBLE', str(int(accuracy_enabled))) + for clip_grad in (0.0, 0.25, 1.0): + saved_clip_values = [] + args = SimpleNamespace( + clip_grad=clip_grad, + template_meta=SimpleNamespace(template_cls=None), + model_meta=SimpleNamespace(is_multimodal=False), + mcore_model=None, + output_dir='unused', + get_model_processor=lambda **kwargs: (None, None), + ) + args.save_args = lambda _, actual=args, captured=saved_clip_values: captured.append(actual.clip_grad) + pipeline = sft.MegatronSft(args) + assert pipeline.args.clip_grad == clip_grad + assert saved_clip_values == [clip_grad] + assert pipeline.template.use_megatron + + def test_get_mcore_model_config_does_not_enable_mtp_from_checkpoint(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) From 0beb52c076508cc2dfce4e273141a4cb618d6d9c Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Mon, 14 Sep 2026 11:42:11 +0800 Subject: [PATCH 32/33] Remove unrelated workflow whitespace from GLM52 change --- .github/workflows/alignment_model_accuracy.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/alignment_model_accuracy.yaml b/.github/workflows/alignment_model_accuracy.yaml index 6de447d2e4..76faf444ba 100644 --- a/.github/workflows/alignment_model_accuracy.yaml +++ b/.github/workflows/alignment_model_accuracy.yaml @@ -65,7 +65,7 @@ jobs: wget -q --no-proxy https://paddle-qa.bj.bcebos.com/CodeSync/develop/PaddleFleet.tar --no-check-certificate rm -rf PaddleFleet && tar xf PaddleFleet.tar && rm -rf PaddleFleet.tar cd PaddleFleet && git pull && cd - - + echo "Download ms-swift form https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz" wget -q --no-proxy https://paddle-github-action.bj.bcebos.com/whl/ms-swift.tar.gz --no-check-certificate rm -rf ms-swift && tar zxf ms-swift.tar.gz && rm -rf ms-swift.tar.gz @@ -145,7 +145,7 @@ jobs: export MEGATRON_CORE_WHEEL_PATH=/workspace/megatron_core-0.0.0-cp312-cp312-linux_x86_64.whl export MS_SWIFT_WHEEL_PATH=/workspace/upload/ms_swift-0.0.0-py3-none-any.whl export MCORE_BRIDGE_WHEEL_PATH=/workspace/mcore_bridge-0.0.0-py3-none-any.whl - # The BOS wheels are republished under a fixed 0.0.0 filename + # The BOS wheels are republished under a fixed 0.0.0 filename export UV_SKIP_WHEEL_FILENAME_CHECK=1 for whl in "$PADDLEFLEET_WHEEL_PATH" "$PADDLEFLEET_OPS_WHEEL_PATH" \ From 76aa97faefef8223a293c5af7175f871a4f9c7ae Mon Sep 17 00:00:00 2001 From: Zhan Rongrui Date: Mon, 14 Sep 2026 16:34:43 +0800 Subject: [PATCH 33/33] refactor(glm52): narrow native integration to required entrypoint behavior --- .../Megatron-SWIFT/Command-line-parameters.md | 3 +- .../Megatron-SWIFT/Command-line-parameters.md | 3 +- scripts/dependence/build.sh | 1 + swift/arguments/base_args/data_args.py | 3 - swift/dataset/__init__.py | 2 +- swift/dataset/utils.py | 24 -- swift/megatron/arguments/megatron_args.py | 3 - swift/megatron/init.py | 118 +++++---- swift/megatron/model/utils.py | 4 +- swift/megatron/pipelines/train/sft.py | 3 + swift/megatron/trainers/trainer.py | 236 +----------------- swift/megatron/trainers/utils.py | 2 +- swift/megatron/utils/utils.py | 21 +- swift/pipelines/train/sft.py | 10 +- tests/megatron/test_accuracy_bridge_tp1.py | 185 +++++++------- tests/megatron/test_accuracy_loss_and_norm.py | 2 +- tests/megatron/test_model_config.py | 67 ++--- tests/megatron/test_pretokenized_dataset.py | 66 ----- tests/megatron/test_raw_loss_observability.py | 35 +-- 19 files changed, 198 insertions(+), 590 deletions(-) delete mode 100644 tests/megatron/test_pretokenized_dataset.py diff --git a/docs/source/Megatron-SWIFT/Command-line-parameters.md b/docs/source/Megatron-SWIFT/Command-line-parameters.md index 6a4be0886c..92f2688400 100644 --- a/docs/source/Megatron-SWIFT/Command-line-parameters.md +++ b/docs/source/Megatron-SWIFT/Command-line-parameters.md @@ -225,7 +225,8 @@ mHC 模块以在支持的 GPU 上获得更好的性能。需要安装 cuTile; - mhc_recompute_layer_num: 每个 MHC 重计算块的层数。设置后,每 `mhc_recompute_layer_num` 层构成一个重计算块。若为 None,Transformer 块中的所有层共享单个重计算块。默认为None。 **MTP参数** -- mtp_num_layers: 多token预测(MTP)层的数量。MTP将每个位置的预测范围扩展到多个未来token。此MTP实现使用D个顺序模块依次预测D个额外的token。默认从config.json中的`num_nextn_predict_layers`或`mtp_num_hidden_layers`读取;命令行显式设置时优先使用命令行值。使用mcore-bridge时,将优先从safetensors文件中加载MTP权重,若无法找到,则进行随机初始化。 +- mtp_num_layers: 多token预测(MTP)层的数量。MTP将每个位置的预测范围扩展到多个未来token。此MTP实现使用D个顺序模块依次预测D个额外的token。默认为None。 + - 注意:mtp_num_layers的值,将不自动从config.json获取,需手动设置。你可以参考config.json中的`num_nextn_predict_layers`, `mtp_num_hidden_layers`字段填写该值。使用mcore-bridge时,将优先从safetensors文件中加载MTP权重,若无法找到,则进行随机初始化。 - mtp_loss_scaling_factor: 多token预测(MTP)损失的缩放因子。我们计算所有深度上MTP损失的平均值,然后乘以该缩放因子得到总体MTP损失,它将作为一个额外的训练目标。默认为0.1。 - mtp_decoder_input_detach: 用来控制 MTP 分支里的 decoder_input 是否停止梯度。默认为False。开启后,MTP loss 不会直接通过 decoder_input 回传到 embedding/vit,但仍会通过 hidden_states 路径更新主干。 - mtp_shared_weights: MTP层之间共享权重,采用GLM-5使用的mtp方案。默认为False。例如你可以设置`--mtp_num_layers 3 --mtp_shared_weights true`。 diff --git a/docs/source_en/Megatron-SWIFT/Command-line-parameters.md b/docs/source_en/Megatron-SWIFT/Command-line-parameters.md index 3771a16d34..984b2f180b 100644 --- a/docs/source_en/Megatron-SWIFT/Command-line-parameters.md +++ b/docs/source_en/Megatron-SWIFT/Command-line-parameters.md @@ -236,7 +236,8 @@ For guidance on selecting parallelization strategies, please refer to the [Train - mhc_recompute_layer_num: Number of layers per MHC recompute block. When set, every `mhc_recompute_layer_num` layers form a recompute block. If `None`, all layers in the transformer block share a single recompute block. Defaults to `None`. **MTP Parameters** -- mtp_num_layers: Number of Multi-Token Prediction (MTP) layers. MTP extends the prediction scope at each position to multiple future tokens. This MTP implementation uses D sequential modules to sequentially predict D additional tokens. By default, the value is read from `num_nextn_predict_layers` or `mtp_num_hidden_layers` in config.json; an explicit command-line value takes precedence. When using mcore-bridge, MTP weights will be loaded from safetensors files first. If not found, random initialization will be performed. +- mtp_num_layers: Number of Multi-Token Prediction (MTP) layers. MTP extends the prediction scope at each position to multiple future tokens. This MTP implementation uses D sequential modules to sequentially predict D additional tokens. Default is None. + - Note: The value of mtp_num_layers will not be automatically retrieved from config.json and must be set manually. You can refer to the `num_nextn_predict_layers`, `mtp_num_hidden_layers` field in config.json to fill in this value. When using mcore-bridge, MTP weights will be loaded from safetensors files first. If not found, random initialization will be performed. - mtp_loss_scaling_factor: Scaling factor of Multi-Token Prediction (MTP) loss. We compute the average of MTP losses across all depths, then multiply it by this scaling factor to obtain the overall MTP loss, which serves as an additional training objective. Default is 0.1. - mtp_decoder_input_detach: Controls whether to stop gradients through decoder_input in the MTP branch. Defaults to False. When enabled, the MTP loss will not back-propagate directly through decoder_input to the embedding/ViT, but will still update the backbone via the hidden_states pathway. - mtp_shared_weights: Share weights across MTP layers, following the MTP scheme proposed in GLM-5. Defaults to False. For example, you can set `--mtp_num_layers 3 --mtp_shared_weights true`. diff --git a/scripts/dependence/build.sh b/scripts/dependence/build.sh index 6b9129513c..de73702692 100644 --- a/scripts/dependence/build.sh +++ b/scripts/dependence/build.sh @@ -79,3 +79,4 @@ echo -e "\033[32m ---- make ms-swift.tar.gz \033[0m" swift_tar echo -e "\033[32m ---- build ms-swift whl \033[0m" swift_build + diff --git a/swift/arguments/base_args/data_args.py b/swift/arguments/base_args/data_args.py index bb79cff2fb..a28ffa6dcb 100644 --- a/swift/arguments/base_args/data_args.py +++ b/swift/arguments/base_args/data_args.py @@ -78,7 +78,6 @@ class DataArguments: val_dataset: List[str] = field(default_factory=list) cached_dataset: List[str] = field(default_factory=list) cached_val_dataset: List[str] = field(default_factory=list) - pretokenized_dataset: bool = False tokenizer_name_or_path: Optional[str] = None split_dataset_ratio: float = 0. @@ -123,8 +122,6 @@ def __post_init__(self): self._init_custom_dataset_info() if isinstance(self.cached_dataset, str): self.cached_dataset = [self.cached_dataset] - if self.pretokenized_dataset and not self.cached_dataset: - raise ValueError('pretokenized_dataset requires cached_dataset') self._init_val_dataset_exists() def _init_val_dataset_exists(self): diff --git a/swift/dataset/__init__.py b/swift/dataset/__init__.py index 876f2c00cc..baa76e0b62 100644 --- a/swift/dataset/__init__.py +++ b/swift/dataset/__init__.py @@ -11,7 +11,7 @@ from .register import (DATASET_MAPPING, DatasetMeta, SubsetDataset, get_dataset_list, register_dataset, register_dataset_info) from .utils import (AddLengthPreprocessor, EncodePreprocessor, LazyLLMDataset, get_temporary_cache_files_directory, - sample_dataset, validate_pretokenized_dataset) + sample_dataset) datasets.fingerprint.get_temporary_cache_files_directory = get_temporary_cache_files_directory datasets.arrow_dataset.get_temporary_cache_files_directory = get_temporary_cache_files_directory diff --git a/swift/dataset/utils.py b/swift/dataset/utils.py index 23e2ac692a..e78abf12f2 100644 --- a/swift/dataset/utils.py +++ b/swift/dataset/utils.py @@ -54,30 +54,6 @@ def sample_dataset( return dataset -def validate_pretokenized_dataset(dataset: HfDataset, max_length: int) -> None: - """Validate fixed token rows without invoking a template or tokenizer.""" - required = {'input_ids', 'labels', 'position_ids', 'lengths'} - if not isinstance(dataset, HfDataset): - raise TypeError('pretokenized_dataset requires a Hugging Face Dataset') - missing = required.difference(dataset.column_names) - if missing: - raise ValueError(f'pretokenized dataset missing columns: {sorted(missing)}') - for row in dataset: - input_ids = row['input_ids'] - labels = row['labels'] - position_ids = row['position_ids'] - length_value = row['lengths'] - length = max(length_value) if isinstance(length_value, list) else length_value - if not isinstance(length, int) or length <= 0 or length > max_length: - raise ValueError(f'pretokenized dataset has invalid length: {length}') - if len(input_ids) != length or len(labels) != length: - raise ValueError('pretokenized dataset has inconsistent input/label length') - if len(position_ids) != length: - raise ValueError('pretokenized dataset has inconsistent position_ids length') - if any(not isinstance(token, int) for token in input_ids + labels + position_ids): - raise TypeError('pretokenized dataset fields must contain integer token values') - - class LazyLLMDataset(Dataset): """This class if used to lazy tokenize the dataset, and skips bad ones when training""" diff --git a/swift/megatron/arguments/megatron_args.py b/swift/megatron/arguments/megatron_args.py index 2af6fdb0c7..0ff44c3efa 100644 --- a/swift/megatron/arguments/megatron_args.py +++ b/swift/megatron/arguments/megatron_args.py @@ -669,7 +669,6 @@ class MegatronArguments(RLHFMegatronArgumentsMixin, MegatronTunerMixin): # mtp mtp_num_layers: Optional[int] = None - num_nextn_predict_layers: Optional[int] = None mtp_loss_scaling_factor: float = 0.1 mtp_decoder_input_detach: bool = False mtp_shared_weights: bool = False @@ -800,8 +799,6 @@ def __post_init__(self): logger.warning(f'Failed to sync dummy template suffix for use_accuracy_compatible: {e}') self._check_mcore_bridge() - if self.mtp_num_layers is None and self.num_nextn_predict_layers is not None: - self.mtp_num_layers = self.num_nextn_predict_layers if self.recompute_granularity == 'none': self.recompute_granularity = None diff --git a/swift/megatron/init.py b/swift/megatron/init.py index 7953834558..4a5e173af2 100644 --- a/swift/megatron/init.py +++ b/swift/megatron/init.py @@ -42,7 +42,6 @@ def _batched_p2p_ops(**kwargs): def _patch_torch_FileSystemReader(): from torch.distributed.checkpoint.filesystem import FileSystemReader from torch.futures import Future - _origin_read_data = FileSystemReader.read_data _origin__slice_file = FileSystemReader._slice_file READER_MAX_WORKERS = int(os.environ.get('MCORE_READER_MAX_WORKERS', '16')) @@ -92,11 +91,11 @@ def _patch_validate_non_overlapping_shards_metadata(): def validate_non_overlapping_shards_metadata(*args, **kwargs): pass - api.validate_non_overlapping_shards_metadata = (validate_non_overlapping_shards_metadata) - api2.validate_non_overlapping_shards_metadata = (validate_non_overlapping_shards_metadata) + api.validate_non_overlapping_shards_metadata = validate_non_overlapping_shards_metadata + api2.validate_non_overlapping_shards_metadata = validate_non_overlapping_shards_metadata def _validate_global_plan(*args, **kwargs): - # torch returns a list of error messages here; empty list means 'no error'. + # torch returns a list of error messages here; empty list means "no error". return [] default_planner._validate_global_plan = _validate_global_plan @@ -125,7 +124,6 @@ def _patch_unified_memory(): return from torch.utils import cpp_extension - load_inline = cpp_extension.load_inline def _new_load_inline(*args, **kwargs): @@ -185,7 +183,6 @@ def _patch_mcore_bridge_disable_te(): import mcore_bridge.model.register as mcb_register def _force_local_spec(orig): - def wrapper(*args, **kwargs): kwargs['use_transformer_engine'] = False return orig(*args, **kwargs) @@ -201,7 +198,11 @@ def wrapper(*args, **kwargs): # Torch DSA TE contaminated post_attn_norm. Force LocalSpecProvider instead. from megatron.core.models.gpt import experimental_attention_variant_module_specs as _eav + origin_backend_spec_provider = _eav._get_backend_spec_provider + def _local_backend_spec_provider(config): + if not getattr(config, 'dsa_accuracy_compatible', False): + return origin_backend_spec_provider(config) from megatron.core.models.backends import LocalSpecProvider return LocalSpecProvider() @@ -242,68 +243,75 @@ def _set_layer_attn(self, mg_layer, hf_state_dict, layer_idx, to_mcore): McbGPTBridge._set_layer_attn = _set_layer_attn - # 4) local-spec dense-MLP norm key mapping: with the local (non-TE) spec the - # dense MLP is `ColumnParallelLinear` + separate `pre_mlp_layernorm` - # (no fused `linear_fc1.layer_norm_weight`); route that key accordingly. - def _set_layer_mlp(self, mg_layer, hf_state_dict, layer_idx, to_mcore, is_mtp=False): - mg_mlp = None if mg_layer is None else mg_layer.mlp - is_moe = True if mg_mlp is not None and hasattr(mg_mlp, 'experts') else False - if not to_mcore: - is_moe = torch.tensor([is_moe], dtype=torch.bool, device='cuda') - if self.pp_size > 1: - dist.all_reduce(is_moe, group=self.pp_group) - if is_moe: - hf_state_dict.update( - self._set_moe_state( - mg_mlp, hf_state_dict, f'{self.hf_mlp_prefix}.', layer_idx, to_mcore, is_mtp=is_mtp)) - self._set_state_dict(mg_layer, 'pre_mlp_layernorm.weight', hf_state_dict, - self.hf_post_attention_layernorm_key, to_mcore) - else: - hf_state_dict.update( - self._set_mlp_state(mg_mlp, hf_state_dict, f'{self.hf_mlp_prefix}.', layer_idx, to_mcore)) - mg_fc1 = None if mg_layer is None else getattr(getattr(mg_layer, 'mlp', None), 'linear_fc1', None) - fused_norm_weight = getattr(mg_fc1, 'layer_norm_weight', None) - if fused_norm_weight is None: - self._set_state_dict(mg_layer, 'pre_mlp_layernorm.weight', hf_state_dict, - self.hf_post_attention_layernorm_key, to_mcore) - else: - self._set_state_dict(mg_layer, 'mlp.linear_fc1.layer_norm_weight', hf_state_dict, - self.hf_post_attention_layernorm_key, to_mcore) - return hf_state_dict + # Dense local MLPs store their norm separately from linear_fc1. Keep the + # bridge's load/export logic, correcting only this TE-specific parameter key. + origin_set_state_dict = McbGPTBridge._set_state_dict + + def _set_state_dict(self, mg_module, mg_key, hf_state_dict, hf_key, to_mcore, **kwargs): + if mg_key == 'mlp.linear_fc1.layer_norm_weight': + fc1 = getattr(getattr(mg_module, 'mlp', None), 'linear_fc1', None) + if getattr(fc1, 'layer_norm_weight', None) is None: + mg_key = 'pre_mlp_layernorm.weight' + return origin_set_state_dict(self, mg_module, mg_key, hf_state_dict, hf_key, to_mcore, **kwargs) - McbGPTBridge._set_layer_mlp = _set_layer_mlp + McbGPTBridge._set_state_dict = _set_state_dict logger.info( 'mcore_bridge patched for TE-off alignment (local spec, persist_layer_norm=False, input_layernorm+mlp-norm map)' ) def _patch_mcore_bridge_tp1_accuracy(): - """Keep the TP1 accuracy graph free of bridge-only viewless nodes.""" + """Apply the DSA TP1 graph choice only within the configured bridge instance.""" + from contextvars import ContextVar + from functools import wraps from mcore_bridge.model.modules import mtp_layer, transformer_block - from megatron.core import parallel_state - def patch_module(module): + active = ContextVar('swift_dsa_tp1_accuracy', default=False) + + def scoped(method): + + @wraps(method) + def call(self, *args, **kwargs): + config = self.config + token = active.set( + getattr(config, 'dsa_accuracy_compatible', False) and config.tensor_model_parallel_size <= 1) + try: + return method(self, *args, **kwargs) + finally: + active.reset(token) + + return call + + def patch_module(module, cls, methods): original = module.make_viewless_tensor if getattr(original, '_swift_tp1_accuracy_patch', False): return def make_viewless_tensor(inp, requires_grad, keep_graph): - if (_use_accuracy_compatible_enabled() and parallel_state.get_tensor_model_parallel_world_size() <= 1): + if active.get(): return inp return original(inp=inp, requires_grad=requires_grad, keep_graph=keep_graph) make_viewless_tensor._swift_tp1_accuracy_patch = True module.make_viewless_tensor = make_viewless_tensor + if hasattr(module, 'gather_from_tensor_model_parallel_region'): + original_gather = module.gather_from_tensor_model_parallel_region + + def gather(input_, group=None): + if active.get() and (group is None or group.size() <= 1): + return input_ + return original_gather(input_, group=group) - patch_module(mtp_layer) - patch_module(transformer_block) + module.gather_from_tensor_model_parallel_region = gather + for name in methods: + setattr(cls, name, scoped(getattr(cls, name))) + + patch_module(mtp_layer, mtp_layer.MultiTokenPredictionLayer, ('_concat_embeddings', '_get_embeddings')) + patch_module(transformer_block, transformer_block.TransformerBlock, ('forward', )) def _patch_mcore_bridge(): - require_version( - 'mcore-bridge>=1.4.0', - 'please install mcore-bridge via `pip install mcore-bridge -U`', - ) + require_version('mcore-bridge>=1.4.0', 'please install mcore-bridge via `pip install mcore-bridge -U`') import mcore_bridge from mcore_bridge import GPTBridge from mcore_bridge.model.register import ModelLoader @@ -327,7 +335,7 @@ def replace_spec_dsa(self, layer_spec): 'indexer', None, ) - if (_use_accuracy_compatible_enabled() and indexer is not None + if (getattr(self.config, 'norm_accuracy_compatible', False) and indexer is not None and getattr(indexer, 'submodules', None) is not None): indexer.submodules.k_norm = WrappedTorchNorm @@ -344,13 +352,7 @@ def save_weights( args=None, processor=None, ) -> None: - origin_save_weights( - self, - mg_models, - output_dir, - peft_format=peft_format, - max_shard_size=max_shard_size, - ) + origin_save_weights(self, mg_models, output_dir, peft_format=peft_format, max_shard_size=max_shard_size) if processor is None or args is None: return hf_config = self.config.hf_config @@ -381,8 +383,7 @@ def save_weights( freeze_vit=args.freeze_vit, freeze_aligner=args.freeze_aligner, include_embedding='all-embedding' in args.target_modules, - exclude_router='all-router' not in args.target_modules, - ) + exclude_router='all-router' not in args.target_modules) else: assert not isinstance(peft_config.target_modules, str), ( 'target_regex is not currently supported for LoRA conversion. Please set `--merge_lora true`.') @@ -401,9 +402,8 @@ def save_weights( llm_config.num_nextn_predict_layers = config.mtp_num_layers HfConfigFactory.del_config_attr(hf_config, 'quantization_config') expert_dtype = None - if (config.fp8 is not None and config.fp8_recipe == 'blockwise' and config.fp8_param): + if config.fp8 is not None and config.fp8_recipe == 'blockwise' and config.fp8_param: from transformers.utils.quantization_config import FineGrainedFP8Config - modules_to_not_convert = get_modules_to_not_convert(self.hf_model) if hasattr(self, '_fp8_skip_modules'): modules_to_not_convert = (modules_to_not_convert or []) + list(self._fp8_skip_modules) @@ -422,8 +422,7 @@ def save_weights( processor, output_dir, model_dirs=[args.model_dir], - additional_saved_files=self.hf_model.model_meta.additional_saved_files, - ) + additional_saved_files=self.hf_model.model_meta.additional_saved_files) logger.info(f'Successfully saved `safetensors` model weights in `{output_dir}`.') dist.barrier() # Ensure all weights are saved completely @@ -448,5 +447,4 @@ def init_megatron_env(): logger.warning('Patch validate_non_overlapping_shards_metadata failed.') pass import megatron.core - logger.info(f'megatron.core.__version__: {megatron.core.__version__}') diff --git a/swift/megatron/model/utils.py b/swift/megatron/model/utils.py index e9ba83267a..96cac5eb78 100644 --- a/swift/megatron/model/utils.py +++ b/swift/megatron/model/utils.py @@ -50,8 +50,8 @@ def get_mcore_model_config(args, hf_config): n_routed_experts = getattr(llm_config, 'n_routed_experts', None) if getattr(llm_config, 'model_type', None) == 'glm_moe_dsa': kwargs['accuracy_compatible_loss_sum_dtype'] = 'float32' - if n_routed_experts is not None: - kwargs['num_moe_experts'] = n_routed_experts + if n_routed_experts is not None: + kwargs['num_moe_experts'] = n_routed_experts # Checkpoint MTP metadata describes available weights, not an opt-in to # auxiliary training. The explicit mtp_num_layers argument below controls it. kwargs['mcore_model_type'] = args.megatron_model_meta.model_type diff --git a/swift/megatron/pipelines/train/sft.py b/swift/megatron/pipelines/train/sft.py index 9b007a2f14..381a02b24d 100644 --- a/swift/megatron/pipelines/train/sft.py +++ b/swift/megatron/pipelines/train/sft.py @@ -43,6 +43,9 @@ def __init__(self, args: Optional[Union[List[str], MegatronSftArguments]] = None self.train_msg = {} super(SwiftSft, self).__init__(args) args = self.args + from megatron.core.transformer.module import _use_accuracy_compatible + if _use_accuracy_compatible() and getattr(args, 'clip_grad', 0) > 0: + args.clip_grad = 0.0 if repatch is not None: megatron_args = asdict(self.args) if args.attention_backend != 'local': diff --git a/swift/megatron/trainers/trainer.py b/swift/megatron/trainers/trainer.py index 035bbd36a4..603fafead1 100644 --- a/swift/megatron/trainers/trainer.py +++ b/swift/megatron/trainers/trainer.py @@ -38,234 +38,7 @@ def project_owning_loader_semantics(input_values, model_label_values, semantic_l class MegatronTrainer(BaseMegatronTrainer): - _LAYER0_FINE_FORWARD_MODULES = { - 'decoder.layers.0.input_layernorm': 'layer0_input_rmsnorm_output', - 'decoder.layers.0.self_attention.linear_q_down_proj': 'layer0_q_down_projection_output', - 'decoder.layers.0.self_attention.q_layernorm': 'layer0_q_rmsnorm_output', - 'decoder.layers.0.self_attention.linear_q_up_proj': 'layer0_q_up_projection_output', - 'decoder.layers.0.self_attention.linear_kv_down_proj': 'layer0_kv_down_projection_output', - 'decoder.layers.0.self_attention.kv_layernorm': 'layer0_kv_rmsnorm_output', - 'decoder.layers.0.self_attention.linear_kv_up_proj': 'layer0_kv_up_projection_output', - 'decoder.layers.0.self_attention.linear_proj': 'layer0_attention_output_projection', - 'decoder.layers.0.self_attention': 'layer0_self_attention_output', - 'decoder.layers.0.mlp.linear_fc1': 'layer0_dense_fc1_output', - 'decoder.layers.0.mlp.linear_fc2': 'layer0_dense_fc2_output', - 'decoder.layers.0.mlp': 'layer0_dense_mlp_output', - 'decoder.layers.0': 'base_transformer_layer_0_output', - } - - @classmethod - def _forward_contract_specs(cls, boundary_set): - if boundary_set == 'coarse': - return None - if boundary_set == 'layer0_fine': - return dict(cls._LAYER0_FINE_FORWARD_MODULES) - raise ValueError(f'unsupported MODEL_REPRO_FORWARD_BOUNDARY_SET: {boundary_set}') - - @staticmethod - def _parameter_record(param): - if hasattr(param, 'is_dist') and param.is_dist(): - param = param._local_value() - tensor = param.detach().contiguous().to(device='cpu') - - def raw_digest(value): - return hashlib.sha256(value.contiguous().view(torch.uint8).numpy().tobytes()).hexdigest() - - zero_count = 0 - negative_zero_count = 0 - if tensor.is_floating_point(): - zero_mask = tensor == 0 - zero_count = int(zero_mask.sum().item()) - negative_zero_count = int((zero_mask & torch.signbit(tensor)).sum().item()) - record = { - 'shape': list(tensor.shape), - 'dtype': str(tensor.dtype), - 'numel': tensor.numel(), - 'sha256': raw_digest(tensor), - 'positive_zero_count': zero_count - negative_zero_count, - 'negative_zero_count': negative_zero_count, - } - if tensor.ndim == 2: - record['transpose_sha256'] = raw_digest(tensor.transpose(0, 1)) - return record - - @staticmethod - def _first_tensor(value): - if isinstance(value, torch.Tensor): - return value - if isinstance(value, dict): - for item in value.values(): - tensor = MegatronTrainer._first_tensor(item) - if tensor is not None: - return tensor - if isinstance(value, (tuple, list)): - for item in value: - tensor = MegatronTrainer._first_tensor(item) - if tensor is not None: - return tensor - return None - - def _write_forward_record(self, boundary, value): - tensor = self._first_tensor(value) - if tensor is None: - return - output_dir = os.environ.get('MODEL_REPRO_FORWARD_RECEIPT_DIR') - rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 - rank_dir = os.path.join(output_dir, f'rank{rank}') - os.makedirs(rank_dir, exist_ok=True) - records = getattr(self, '_forward_contract_records', {}) - call_index = sum(name == boundary or name.startswith(f'{boundary}_call') for name in records) - name = boundary if call_index == 0 else f'{boundary}_call{call_index}' - tensor = tensor.detach().contiguous().to(device='cpu') - raw = tensor.view(torch.uint8).numpy().tobytes() - file_name = ''.join(character if character.isalnum() or character in '-_' else '_' for character in name) - raw_path = os.path.join(rank_dir, f'{file_name}.bin') - with open(raw_path, 'wb') as stream: - stream.write(raw) - zero_count = 0 - negative_zero_count = 0 - if tensor.is_floating_point(): - zero_mask = tensor == 0 - zero_count = int(zero_mask.sum().item()) - negative_zero_count = int((zero_mask & torch.signbit(tensor)).sum().item()) - records[name] = { - 'boundary': boundary, - 'shape': list(tensor.shape), - 'dtype': str(tensor.dtype), - 'numel': tensor.numel(), - 'sha256': hashlib.sha256(raw).hexdigest(), - 'positive_zero_count': zero_count - negative_zero_count, - 'negative_zero_count': negative_zero_count, - 'raw_path': raw_path, - } - self._forward_contract_records = records - payload = { - 'schema': 'glm52-local-forward-boundaries/v1', - 'framework': 'torch', - 'rank': rank, - 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, - 'boundary_set': getattr(self, '_forward_contract_boundary_set', 'coarse'), - 'selectors': getattr(self, '_forward_contract_selector_receipt', []), - 'records': records, - } - with open(os.path.join(rank_dir, 'metadata.json'), 'w', encoding='utf-8') as stream: - json.dump(payload, stream, ensure_ascii=False, indent=2, sort_keys=True) - stream.write('\n') - - def _install_forward_contract_once(self): - output_dir = os.environ.get('MODEL_REPRO_FORWARD_RECEIPT_DIR') - if not output_dir or getattr(self, '_forward_contract_installed', False): - return - rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 - boundary_set = os.environ.get('MODEL_REPRO_FORWARD_BOUNDARY_SET', 'coarse') - fine_specs = self._forward_contract_specs(boundary_set) - self._forward_contract_boundary_set = boundary_set - handles = [] - if fine_specs is not None: - selected = [] - if rank < 2: - module_hits = {name: [] for name in fine_specs} - for chunk_index, model in enumerate(self.unwrapped_models): - for module_name, module in model.named_modules(): - if module_name in module_hits: - module_hits[module_name].append((chunk_index, module)) - invalid = {name: len(hits) for name, hits in module_hits.items() if len(hits) != 1} - if invalid: - raise RuntimeError( - f'layer0 fine forward selectors must match exactly once on rank {rank}: {invalid}') - for module_name, boundary in fine_specs.items(): - chunk_index, module = module_hits[module_name][0] - handles.append( - module.register_forward_hook( - lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) - selected.append({'chunk': chunk_index, 'module': module_name, 'boundary': boundary}) - self._forward_contract_selector_receipt = selected - rank_dir = os.path.join(output_dir, f'rank{rank}') - os.makedirs(rank_dir, exist_ok=True) - with open(os.path.join(rank_dir, 'metadata.json'), 'w', encoding='utf-8') as stream: - json.dump( - { - 'schema': 'glm52-local-forward-boundaries/v1', - 'framework': 'torch', - 'rank': rank, - 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, - 'boundary_set': boundary_set, - 'selectors': selected, - 'records': {}, - }, - stream, - ensure_ascii=False, - indent=2, - sort_keys=True) - stream.write('\n') - self._forward_contract_handles = handles - self._forward_contract_installed = True - return - for chunk_index, model in enumerate(self.unwrapped_models): - for module_name, module in model.named_modules(): - boundary = None - match = __import__('re').fullmatch(r'decoder\.layers\.(\d+)', module_name) - if module_name == 'embedding': - boundary = f'chunk{chunk_index}_embedding_output' - elif match: - local_layer = int(match.group(1)) - global_layer = local_layer if rank < 2 else local_layer + 2 - input_boundary = f'base_layer_{global_layer}_input' - handles.append( - module.register_forward_pre_hook( - lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) - boundary = f'base_layer_{global_layer}_output' - elif module_name == 'decoder.final_layernorm': - boundary = 'final_norm_output' - elif module_name == 'output_layer': - input_boundary = 'output_head_input' - handles.append( - module.register_forward_pre_hook( - lambda _module, inputs, name=input_boundary: self._write_forward_record(name, inputs))) - boundary = 'output_head_output' - elif module_name.startswith('mtp.layers.0.') and module_name.rsplit('.', 1)[-1] in { - 'enorm', 'hnorm', 'eh_proj', 'mtp_model_layer', 'layer_norm', 'final_layernorm' - }: - boundary = f"mtp_{module_name.removeprefix('mtp.layers.0.').replace('.', '_')}_output" - if boundary is not None: - handles.append( - module.register_forward_hook( - lambda _module, _inputs, output, name=boundary: self._write_forward_record(name, output))) - self._forward_contract_handles = handles - self._forward_contract_installed = True - - def _write_parameter_contract_once(self): - output_dir = os.environ.get('MODEL_REPRO_PARAMETER_RECEIPT_DIR') - if not output_dir or getattr(self, '_parameter_contract_written', False): - return - rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 - parameters = [] - for chunk_index, model in enumerate(self.unwrapped_models): - for name, param in model.named_parameters(): - parameters.append({ - 'chunk': chunk_index, - 'name': name, - **self._parameter_record(param), - }) - payload = { - 'schema': 'glm52-loaded-parameter-inventory/v1', - 'framework': 'torch', - 'rank': rank, - 'world_size': torch.distributed.get_world_size() if torch.distributed.is_initialized() else 1, - 'parameters': parameters, - 'parameter_count': len(parameters), - 'local_numel': sum(item['numel'] for item in parameters), - } - os.makedirs(output_dir, exist_ok=True) - path = os.path.join(output_dir, f'rank{rank}.json') - with open(path, 'w', encoding='utf-8') as stream: - json.dump(payload, stream, ensure_ascii=False, indent=2, sort_keys=True) - stream.write('\n') - self._parameter_contract_written = True - def _write_input_contract_once(self, data, seq_lens=None): - self._write_parameter_contract_once() - self._install_forward_contract_once() path = os.environ.get('MODEL_REPRO_INPUT_RECEIPT_PATH') if not path or getattr(self, '_input_contract_written', False): return @@ -289,8 +62,7 @@ def digest(items): label_values = values(labels) model_mask_values = [label != -100 for label in label_values] semantic_length = seq_lens[0] if seq_lens else len(input_values) - labels_were_shifted = self.args.task_type == 'causal_lm' and not getattr(self.args, 'pretokenized_dataset', - False) + labels_were_shifted = self.args.task_type == 'causal_lm' semantic_input_values, semantic_label_values, semantic_mask_values = project_owning_loader_semantics( input_values, label_values, semantic_length, labels_were_shifted) payload = { @@ -407,9 +179,9 @@ def loss_func(self, import hashlib as _hashlib _final = (loss[0].detach().float() / loss[1].detach().float().clamp(min=1)).contiguous() print( - f'\nfinal_loss: rank={torch.distributed.get_rank()} ' - f'val={_final.item():.20f} ' - f'md5={_hashlib.md5(_final.cpu().numpy().tobytes()).hexdigest()}', + f"\nfinal_loss: rank={torch.distributed.get_rank()} " + f"val={_final.item():.20f} " + f"md5={_hashlib.md5(_final.cpu().numpy().tobytes()).hexdigest()}", flush=True) metrics = {'loss': reporting_loss} diff --git a/swift/megatron/trainers/utils.py b/swift/megatron/trainers/utils.py index 1817fca687..a77d4e6a4f 100644 --- a/swift/megatron/trainers/utils.py +++ b/swift/megatron/trainers/utils.py @@ -19,7 +19,7 @@ def get_batch_on_this_pp_rank(args, data, vp_stage=None): - if args.task_type == 'causal_lm' and not getattr(args, 'pretokenized_dataset', False): + if args.task_type == 'causal_lm': data['labels'] = torch.roll(data['labels'], -1, dims=-1) if 'loss_scale' in data: data['loss_scale'] = torch.roll(data['loss_scale'], -1, dims=-1) diff --git a/swift/megatron/utils/utils.py b/swift/megatron/utils/utils.py index 4577b06f78..4774c99095 100644 --- a/swift/megatron/utils/utils.py +++ b/swift/megatron/utils/utils.py @@ -212,10 +212,9 @@ def get_padding_to(args): padding_to = None if args.tensor_model_parallel_size > 1 and args.sequence_parallel: padding_to = args.tensor_model_parallel_size - # Sequence-parallel interleaves the sequence across TP ranks at 2x. - # Without this the collator pads to ceil(57/TP)*TP=58 while the - # PaddleFleet carrier and E-811 IEEE 1-100 are 60 (57 -> TP*SP=4). - padding_to = padding_to * 2 + # Match the DSA reference carrier without changing other TP+SP models. + if (getattr(args, 'megatron_extra_kwargs', None) or {}).get('dsa_accuracy_compatible', False): + padding_to *= 2 if args.context_parallel_size > 1: padding_to = (padding_to or 1) * args.context_parallel_size origin_padding_to = padding_to @@ -293,7 +292,7 @@ def get_load_fixed_data_path(): def _batch_data_suffix(step, rank, seq_len): - return f'step{step}_rank{rank}_seq{seq_len}.npy' + return f"step{step}_rank{rank}_seq{seq_len}.npy" def dump_batch_data(batch, step, seq_len): @@ -308,10 +307,10 @@ def dump_batch_data(batch, step, seq_len): torch.cuda.synchronize() os.makedirs(dump_path, exist_ok=True) suffix = _batch_data_suffix(step, rank, seq_len) - np.save(os.path.join(dump_path, f'tokens_{suffix}'), tokens.detach().cpu().numpy()) - np.save(os.path.join(dump_path, f'labels_{suffix}'), labels.detach().cpu().numpy()) + np.save(os.path.join(dump_path, f"tokens_{suffix}"), tokens.detach().cpu().numpy()) + np.save(os.path.join(dump_path, f"labels_{suffix}"), labels.detach().cpu().numpy()) if rank == 0: - print(f'[DUMP_DATA_PATH] saved tokens_{suffix} and labels_{suffix}', flush=True) + print(f"[DUMP_DATA_PATH] saved tokens_{suffix} and labels_{suffix}", flush=True) def load_fixed_batch_data(batch, step, seq_len): @@ -321,11 +320,11 @@ def load_fixed_batch_data(batch, step, seq_len): rank = torch.distributed.get_rank() if torch.distributed.is_initialized() else 0 suffix = _batch_data_suffix(step, rank, seq_len) - tokens_file = os.path.join(load_path, f'tokens_{suffix}') - labels_file = os.path.join(load_path, f'labels_{suffix}') + tokens_file = os.path.join(load_path, f"tokens_{suffix}") + labels_file = os.path.join(load_path, f"labels_{suffix}") if not (os.path.exists(tokens_file) and os.path.exists(labels_file)): if rank == 0: - print(f'[LOAD_FIXED_DATA_PATH] file not found: {tokens_file}', flush=True) + print(f"[LOAD_FIXED_DATA_PATH] file not found: {tokens_file}", flush=True) return batch tokens_np = np.load(tokens_file) diff --git a/swift/pipelines/train/sft.py b/swift/pipelines/train/sft.py index 13eb0f3bcf..8d547f4106 100644 --- a/swift/pipelines/train/sft.py +++ b/swift/pipelines/train/sft.py @@ -5,7 +5,7 @@ from swift.arguments import SftArguments from swift.dataset import (AddLengthPreprocessor, DatasetLoader, EncodePreprocessor, IterablePackingDataset, - LazyLLMDataset, PackingDataset, validate_pretokenized_dataset) + LazyLLMDataset, PackingDataset) from swift.infer_engine import prepare_generation_config from swift.ray_utils import RayHelper from swift.sequence_parallel import sequence_parallel @@ -49,8 +49,6 @@ def _prepare_generation_config(self): def _prepare_model_tokenizer(self, **kwargs): args = self.args self.model, self.processor = args.get_model_processor(**kwargs) - if getattr(args, 'tokenizer_name_or_path', None): - logger.info(f'Using independent tokenizer_name_or_path: {args.tokenizer_name_or_path}') if args.sequence_parallel_size > 1: sequence_parallel.prepare( args.sequence_parallel_size, model=self.model, tokenizer=self.processor, padding_free=args.padding_free) @@ -126,12 +124,6 @@ def _prepare_dataset(self): def _post_process_datasets(self, datasets: List) -> List: args = self.args - if args.pretokenized_dataset: - for dataset in datasets: - if dataset is not None: - validate_pretokenized_dataset(dataset, args.max_length) - logger.info('Using pretokenized cached dataset without template encoding.') - return datasets predict_with_generate = getattr(args, 'predict_with_generate', False) template = self.template diff --git a/tests/megatron/test_accuracy_bridge_tp1.py b/tests/megatron/test_accuracy_bridge_tp1.py index 30a261b219..9c8ed89768 100644 --- a/tests/megatron/test_accuracy_bridge_tp1.py +++ b/tests/megatron/test_accuracy_bridge_tp1.py @@ -1,6 +1,4 @@ -"""Unit tests for ms-swift TP1 accuracy bridge patch.""" -from __future__ import annotations - +"""Instance-scoped TP1 bridge behavior, including mixed configurations.""" import ast import sys import unittest @@ -8,113 +6,98 @@ from types import ModuleType, SimpleNamespace from unittest.mock import patch -ROOT = Path(__file__).resolve().parents[2] -_UAC = {'on': False} -_TP = {'size': 1} - - -def _use_accuracy_compatible_enabled(): - return _UAC['on'] - -def _load_patch(): - src = (ROOT / 'swift/megatron/init.py').read_text() - tree = ast.parse(src) - target = next( - node for node in tree.body +def load_patch(): + path = Path(__file__).resolve().parents[2] / 'swift/megatron/init.py' + node = next( + node for node in ast.parse(path.read_text()).body if isinstance(node, ast.FunctionDef) and node.name == '_patch_mcore_bridge_tp1_accuracy') - target.decorator_list = [] - mod = ast.Module(body=[target], type_ignores=[]) - ast.fix_missing_locations(mod) - ns = {'_use_accuracy_compatible_enabled': _use_accuracy_compatible_enabled} - exec(compile(mod, 'swift/megatron/init.py', 'exec'), ns) - return ns['_patch_mcore_bridge_tp1_accuracy'] - - -_patch_mcore_bridge_tp1_accuracy = _load_patch() - - -def _make_original(name): - - def make_viewless_tensor(inp, requires_grad, keep_graph): - make_viewless_tensor.calls.append((inp, requires_grad, keep_graph)) - return ('wrapped', inp) - - make_viewless_tensor.calls = [] - make_viewless_tensor.__name__ = name - return make_viewless_tensor + namespace = {} + exec(compile(ast.Module(body=[node], type_ignores=[]), str(path), 'exec'), namespace) + return namespace[node.name] class TestBridgeTp1Patch(unittest.TestCase): def setUp(self): - self.orig_mtp = _make_original('mtp') - self.orig_block = _make_original('block') - self.mtp = ModuleType('mtp_layer') - self.block = ModuleType('transformer_block') - self.mtp.make_viewless_tensor = self.orig_mtp - self.block.make_viewless_tensor = self.orig_block - - fake_ps = SimpleNamespace(get_tensor_model_parallel_world_size=lambda: _TP['size']) - fake_core = ModuleType('megatron.core') - fake_core.parallel_state = fake_ps - fake_mcore = ModuleType('mcore_bridge') - fake_model = ModuleType('mcore_bridge.model') - fake_modules = ModuleType('mcore_bridge.model.modules') - fake_modules.mtp_layer = self.mtp - fake_modules.transformer_block = self.block - fake_megatron = ModuleType('megatron') - - p = patch.dict( - sys.modules, - { - 'megatron': fake_megatron, - 'megatron.core': fake_core, - 'mcore_bridge': fake_mcore, - 'mcore_bridge.model': fake_model, - 'mcore_bridge.model.modules': fake_modules, - }, - ) - p.start() - self.addCleanup(p.stop) - - def tearDown(self): - _UAC['on'] = False - _TP['size'] = 1 - - def test_idempotent_and_both_module_globals(self): - _patch_mcore_bridge_tp1_accuracy() - first_mtp = self.mtp.make_viewless_tensor - first_block = self.block.make_viewless_tensor - self.assertTrue(getattr(first_mtp, '_swift_tp1_accuracy_patch', False)) - self.assertTrue(getattr(first_block, '_swift_tp1_accuracy_patch', False)) - _patch_mcore_bridge_tp1_accuracy() - self.assertIs(self.mtp.make_viewless_tensor, first_mtp) - self.assertIs(self.block.make_viewless_tensor, first_block) - - def test_uac_tp1_identity_both_modules(self): - _patch_mcore_bridge_tp1_accuracy() - _UAC['on'] = True - _TP['size'] = 1 + self.modules = [] + for name, class_name in (('mtp_layer', 'MultiTokenPredictionLayer'), ('transformer_block', 'TransformerBlock')): + module = ModuleType(name) + module.make_viewless_tensor = lambda inp, **kwargs: ('wrapped', inp) + module.gather_from_tensor_model_parallel_region = lambda inp, **kwargs: ('gathered', inp) + + def forward(self, inp, callback=None, module=module): + result = module.make_viewless_tensor(inp=inp, requires_grad=True, keep_graph=True) + if callback: + callback() + return result + + cls = type(class_name, (), {'forward': forward, '_concat_embeddings': forward, '_get_embeddings': forward}) + setattr(module, class_name, cls) + self.modules.append((module, cls)) + modules = ModuleType('mcore_bridge.model.modules') + modules.mtp_layer, modules.transformer_block = [item[0] for item in self.modules] + replacement = patch.dict(sys.modules, {'mcore_bridge.model.modules': modules}) + replacement.start() + self.addCleanup(replacement.stop) + self.apply_patch = load_patch() + self.apply_patch() + + def instance(self, cls, enabled, tp_size=1): + instance = cls() + instance.config = SimpleNamespace(dsa_accuracy_compatible=enabled, tensor_model_parallel_size=tp_size) + return instance + + def test_idempotent(self): + originals = [(module.make_viewless_tensor, cls.forward) for module, cls in self.modules] + self.apply_patch() + for (module, cls), (function, method) in zip(self.modules, originals): + self.assertIs(module.make_viewless_tensor, function) + self.assertIs(cls.forward, method) + + def test_only_explicit_dsa_tp1_instance_skips_viewless(self): inp = object() - self.assertIs(self.mtp.make_viewless_tensor(inp, True, True), inp) - self.assertIs(self.block.make_viewless_tensor(inp, False, False), inp) - self.assertEqual(self.orig_mtp.calls, []) - self.assertEqual(self.orig_block.calls, []) - - def test_off_and_actual_tp2_delegate_to_original(self): - _patch_mcore_bridge_tp1_accuracy() + for module, cls in self.modules: + name = '_concat_embeddings' if module.__name__ == 'mtp_layer' else 'forward' + for enabled, tp_size in ((False, 1), (True, 1), (True, 2)): + output = getattr(self.instance(cls, enabled, tp_size), name)(inp) + if enabled and tp_size == 1: + self.assertIs(output, inp) + else: + self.assertEqual(output, ('wrapped', inp)) + self.assertEqual(module.make_viewless_tensor(inp, True, True), ('wrapped', inp)) + + def test_gather_uses_instance_scope_and_actual_group(self): + module, cls = self.modules[1] inp = object() - _UAC['on'] = False - _TP['size'] = 1 - out = self.mtp.make_viewless_tensor(inp, True, True) - self.assertEqual(out, ('wrapped', inp)) - self.assertEqual(self.orig_mtp.calls[-1], (inp, True, True)) - _UAC['on'] = True - _TP['size'] = 2 - out2 = self.block.make_viewless_tensor(inp, False, True) - self.assertEqual(out2, ('wrapped', inp)) - self.assertEqual(self.orig_block.calls[-1], (inp, False, True)) + for enabled, tp_size in ((False, 1), (True, 1), (True, 2)): + observed = [] + group = SimpleNamespace(size=lambda: tp_size) + + def callback(): + observed.append(module.gather_from_tensor_model_parallel_region(inp, group)) + + self.instance(cls, enabled).forward(inp, callback=callback) + if enabled and tp_size == 1: + self.assertIs(observed[0], inp) + else: + self.assertEqual(observed[0], ('gathered', inp)) + self.assertEqual(module.gather_from_tensor_model_parallel_region(inp), ('gathered', inp)) + + def test_nested_legacy_call_and_exception_restore_scope(self): + module, cls = self.modules[1] + enabled = self.instance(cls, True) + legacy = self.instance(cls, False) + inp = object() + + def nested(): + self.assertEqual(legacy.forward(inp), ('wrapped', inp)) + self.assertIs(module.make_viewless_tensor(inp, True, True), inp) + raise ValueError('test failure') + + with self.assertRaisesRegex(ValueError, 'test failure'): + enabled.forward(inp, callback=nested) + self.assertEqual(module.make_viewless_tensor(inp, True, True), ('wrapped', inp)) if __name__ == '__main__': diff --git a/tests/megatron/test_accuracy_loss_and_norm.py b/tests/megatron/test_accuracy_loss_and_norm.py index 57241f7900..235f94ab32 100644 --- a/tests/megatron/test_accuracy_loss_and_norm.py +++ b/tests/megatron/test_accuracy_loss_and_norm.py @@ -100,7 +100,7 @@ def test_indexer_norm_preserves_disabled_provider(self): with patch.dict(sys.modules, {module.__name__: module}): replace_spec(loader, spec) self.assertEqual(calls, ['provider']) - self.assertIs(indexer.submodules.k_norm, native_norm if enabled else provider_norm) + self.assertIs(indexer.submodules.k_norm, native_norm if norm_accuracy else provider_norm) expected_qkv = native_norm if norm_accuracy else provider_norm self.assertIs(attention.submodules.q_layernorm, expected_qkv) self.assertIs(attention.submodules.kv_layernorm, expected_qkv) diff --git a/tests/megatron/test_model_config.py b/tests/megatron/test_model_config.py index 77f6f5315c..bbf85f99d0 100644 --- a/tests/megatron/test_model_config.py +++ b/tests/megatron/test_model_config.py @@ -1,3 +1,4 @@ +import ast import inspect import math import torch @@ -61,37 +62,6 @@ def test_get_mcore_model_config_propagates_accuracy_mode(monkeypatch): assert config.kwargs['use_accuracy_compatible'] is enabled -def test_sft_pipeline_preserves_explicit_gradient_clipping(monkeypatch): - from swift.megatron.pipelines.train import sft - - def initialize_arguments(pipeline, args): - pipeline.args = args - - def prepare_template(pipeline): - pipeline.template = SimpleNamespace() - - monkeypatch.setattr(sft.SwiftSft.__mro__[1], '__init__', initialize_arguments) - monkeypatch.setattr(sft.MegatronSft, '_prepare_template', prepare_template) - monkeypatch.setattr(sft, 'repatch', None) - for accuracy_enabled in (False, True): - monkeypatch.setenv('USE_ACCURACY_COMPATIBLE', str(int(accuracy_enabled))) - for clip_grad in (0.0, 0.25, 1.0): - saved_clip_values = [] - args = SimpleNamespace( - clip_grad=clip_grad, - template_meta=SimpleNamespace(template_cls=None), - model_meta=SimpleNamespace(is_multimodal=False), - mcore_model=None, - output_dir='unused', - get_model_processor=lambda **kwargs: (None, None), - ) - args.save_args = lambda _, actual=args, captured=saved_clip_values: captured.append(actual.clip_grad) - pipeline = sft.MegatronSft(args) - assert pipeline.args.clip_grad == clip_grad - assert saved_clip_values == [clip_grad] - assert pipeline.template.use_megatron - - def test_get_mcore_model_config_does_not_enable_mtp_from_checkpoint(monkeypatch): _patch_model_config(monkeypatch) hf_config = PretrainedConfig(num_nextn_predict_layers=1) @@ -133,7 +103,7 @@ def test_get_mcore_model_config_does_not_enable_mtp_from_nested_checkpoint(monke def test_get_mcore_model_config_prefers_n_routed_experts(monkeypatch): _patch_model_config(monkeypatch) - hf_config = PretrainedConfig(num_experts=256, n_routed_experts=16) + hf_config = PretrainedConfig(model_type="glm_moe_dsa", num_experts=256, n_routed_experts=16) config = utils.get_mcore_model_config(_make_args(), hf_config) @@ -161,7 +131,11 @@ def test_get_padding_to_sequence_parallel_uses_tp_times_two(): fp4=None, attention_backend='unfused', ) + assert get_padding_to(args) == 2 + args.megatron_extra_kwargs = {"dsa_accuracy_compatible": True} assert get_padding_to(args) == 4 + args.megatron_extra_kwargs = {"dsa_accuracy_compatible": False} + assert get_padding_to(args) == 2 seq_len = 57 assert math.ceil(seq_len / 4) * 4 == 60 assert math.ceil(seq_len / 2) * 2 == 58 @@ -185,19 +159,12 @@ def test_dsa_backend_forced_to_local_spec_when_accuracy_compatible(monkeypatch): monkeypatch.setattr(init, '_use_accuracy_compatible_enabled', lambda: True) init._patch_mcore_bridge_disable_te() - provider = eav._get_backend_spec_provider(SimpleNamespace()) + provider = eav._get_backend_spec_provider(SimpleNamespace(dsa_accuracy_compatible=True)) assert isinstance(provider, LocalSpecProvider) assert hasattr(provider, 'linear') assert provider.linear() is not provider.column_parallel_linear() -def test_local_spec_mlp_norm_maps_pre_mlp_layernorm_when_unfused(): - source = inspect.getsource(_patch_mcore_bridge_disable_te) - assert 'fused_norm_weight is None' in source - assert 'pre_mlp_layernorm.weight' in source - assert 'mlp.linear_fc1.layer_norm_weight' in source - - def test_dsa_index_share_rejects_selective_recompute(): config = SimpleNamespace( experimental_attention_variant='dsa', @@ -211,3 +178,23 @@ def test_dsa_index_share_rejects_selective_recompute(): assert 'Set recompute_granularity=none' in str(error) else: raise AssertionError('expected DSA index sharing with selective recompute to fail closed') + + +def test_local_dense_norm_binding_keeps_other_parameter_keys(): + tree = ast.parse(inspect.getsource(_patch_mcore_bridge_disable_te)) + node = next(node for node in ast.walk(tree) if isinstance(node, ast.FunctionDef) and node.name == '_set_state_dict') + namespace = {'origin_set_state_dict': lambda *args, **kwargs: (args, kwargs)} + exec(compile(ast.Module(body=[node], type_ignores=[]), '', 'exec'), namespace) + bind = namespace['_set_state_dict'] + local = SimpleNamespace(mlp=SimpleNamespace(linear_fc1=SimpleNamespace())) + fused = SimpleNamespace(mlp=SimpleNamespace(linear_fc1=SimpleNamespace(layer_norm_weight=object()))) + for layer, key, expected in ( + (local, 'mlp.linear_fc1.layer_norm_weight', 'pre_mlp_layernorm.weight'), + (None, 'mlp.linear_fc1.layer_norm_weight', 'pre_mlp_layernorm.weight'), + (fused, 'mlp.linear_fc1.layer_norm_weight', 'mlp.linear_fc1.layer_norm_weight'), + (local, 'mlp.linear_fc1.weight', 'mlp.linear_fc1.weight'), + ): + args, kwargs = bind(object(), layer, key, {}, 'post_attention_layernorm.weight', True, offset=0.5) + assert args[2] == expected + assert args[4:] == ('post_attention_layernorm.weight', True) + assert kwargs == {'offset': 0.5} diff --git a/tests/megatron/test_pretokenized_dataset.py b/tests/megatron/test_pretokenized_dataset.py deleted file mode 100644 index cc486241d2..0000000000 --- a/tests/megatron/test_pretokenized_dataset.py +++ /dev/null @@ -1,66 +0,0 @@ -import pytest -import torch -from datasets import Dataset -from types import SimpleNamespace - -from swift.dataset import validate_pretokenized_dataset -from swift.megatron.trainers import utils as trainer_utils -from swift.model.register import ModelLoader - - -def test_validate_pretokenized_dataset_accepts_fixed_tokens(): - dataset = Dataset.from_dict({ - 'input_ids': [[154820, 42, 42, 17, 99, 42, 8]], - 'labels': [[42, 42, 17, 99, 42, 8, 3]], - 'position_ids': [[0, 1, 2, 3, 4, 5, 6]], - 'lengths': [7], - }) - - validate_pretokenized_dataset(dataset, max_length=7) - - -def test_validate_pretokenized_dataset_rejects_missing_labels(): - dataset = Dataset.from_dict({ - 'input_ids': [[1, 2]], - 'position_ids': [[0, 1]], - 'lengths': [2], - }) - - with pytest.raises(ValueError, match='missing columns'): - validate_pretokenized_dataset(dataset, max_length=2) - - -def test_pretokenized_labels_are_not_shifted(monkeypatch): - labels = torch.tensor([[42, 42, 17, 99, 42, 8, 3]]) - data = {'input_ids': torch.zeros_like(labels), 'labels': labels.clone()} - args = SimpleNamespace( - task_type='causal_lm', - pretokenized_dataset=True, - pipeline_model_parallel_size=1, - ) - monkeypatch.setattr(trainer_utils, 'get_current_device', lambda: 'cpu') - monkeypatch.setattr(trainer_utils, 'to_device', lambda value, *_args, **_kwargs: value) - - batch = trainer_utils.get_batch_on_this_pp_rank(args, data) - - assert torch.equal(batch['labels'], labels) - - -def test_model_loader_uses_independent_processor_path(): - requested = [] - - class FakeTokenizer: - - @classmethod - def from_pretrained(cls, path, **kwargs): - requested.append((path, kwargs)) - return object() - - loader = ModelLoader.__new__(ModelLoader) - loader.processor_id_or_path = '/tokenizer-only' - loader.auto_tokenizer_cls = FakeTokenizer - loader.default_trust_remote_code = True - - loader.get_processor('/weights-only', config=None) - - assert requested == [('/tokenizer-only', {'trust_remote_code': True})] diff --git a/tests/megatron/test_raw_loss_observability.py b/tests/megatron/test_raw_loss_observability.py index 6de59899f1..f7fd6b3485 100644 --- a/tests/megatron/test_raw_loss_observability.py +++ b/tests/megatron/test_raw_loss_observability.py @@ -1,7 +1,5 @@ -import torch - from swift.megatron.callbacks.print import raw_loss_event -from swift.megatron.trainers.trainer import MegatronTrainer, project_owning_loader_semantics +from swift.megatron.trainers.trainer import project_owning_loader_semantics def test_raw_loss_event_preserves_unrounded_values_and_step(): @@ -39,34 +37,3 @@ def test_owning_loader_projection_rejects_invalid_semantic_length(): assert 'invalid owning-loader semantic length' in str(exc) else: raise AssertionError('invalid semantic length was accepted') - - -def test_parameter_record_preserves_orientation_and_signed_zero(): - tensor = torch.tensor([[0.0, -0.0], [1.0, 2.0]], dtype=torch.float32) - record = MegatronTrainer._parameter_record(tensor) - assert record['shape'] == [2, 2] - assert record['dtype'] == 'torch.float32' - assert record['positive_zero_count'] == 1 - assert record['negative_zero_count'] == 1 - assert record['sha256'] != record['transpose_sha256'] - - -def test_layer0_fine_forward_specs_are_explicit_and_fail_closed(): - assert MegatronTrainer._forward_contract_specs('coarse') is None - specs = MegatronTrainer._forward_contract_specs('layer0_fine') - assert len(specs) == 13 - assert specs['decoder.layers.0.input_layernorm'] == 'layer0_input_rmsnorm_output' - assert specs['decoder.layers.0.self_attention.linear_q_down_proj'] == 'layer0_q_down_projection_output' - assert specs['decoder.layers.0'] == 'base_transformer_layer_0_output' - try: - MegatronTrainer._forward_contract_specs('unknown') - except ValueError as exc: - assert 'unsupported MODEL_REPRO_FORWARD_BOUNDARY_SET' in str(exc) - else: - raise AssertionError('unknown forward boundary set was accepted') - - -def test_first_tensor_prefers_first_tensor_in_nested_module_output(): - first = torch.tensor([1.0]) - second = torch.tensor([2.0]) - assert MegatronTrainer._first_tensor((None, {'output': first}, second)) is first