Natural Memory NM2.1: 记忆路由器分叉、数据集缺陷修复与全轴评测证据

- 引入 MemoryRouterXL 与 v5/v6 流式多线程训练/编码管线
- 修复 prepare_memory_router_dataset 候选池重建缺陷(mega 家族 3568x 加速,输出逐字节相同)
- 修复 v5 被破坏的拒答与多跳标签(train 未知样本 319 -> 16319,multi_hop 平均正例 1.00 -> 2.00)
- 同存储预算下 V2-128 v6 逐轴 22/22 通过:Top-1 41.12% -> 94.62%,未知拒答 0.00% -> 100.00%
- 记录三条被实测推翻的显然优化(logits_to_keep=1 反而慢 55%、XL 容量未带来收益)
- 记忆手术跨架构可移植性 14/14,读写关闭时与原生模型逐位相同
This commit is contained in:
WpyQwq
2026-09-19 11:11:31 +08:00
commit 643e22ecb9
484 changed files with 306821 additions and 0 deletions
+80
View File
@@ -0,0 +1,80 @@
[
{
"model": "hf-internal-testing/tiny-random-LlamaForCausalLM",
"full_stack": false,
"stages": {
"1_config": "needs patch: config.text_config",
"2_load_causal_lm": "ok",
"3_container": "needs patch: repo wants model.language_model.layers, found model.layers",
"4_convention": "MISMATCH: second positional is 'attention_mask', adapter forwards position_embeddings positionally",
"5_surgery": "FAIL ValueError: layer_indices must be inside [0, 2)"
},
"patches": [
"config.text_config = config",
"model.model.language_model = model.model"
],
"architecture": "LlamaForCausalLM",
"model_type": "llama",
"hidden_size": 16,
"num_layers": 2,
"vocab_size": 32000,
"text_config_present": false,
"backbone_class": "LlamaForCausalLM",
"container_path": "model.layers",
"repo_layer_path_ok": false,
"layer_forward_params": [
"hidden_states",
"attention_mask",
"position_ids",
"past_key_values",
"use_cache",
"position_embeddings",
"kwargs"
],
"second_positional_is_position_embeddings": false,
"traceback": "Traceback (most recent call last):\n File \"H:\\Memory\\V2_dpskw\\probe_nm2_portability.py\", line 219, in probe\n layer_indices = memory_config.resolved_layers(num_layers)\n ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^\n File \"H:\\Memory\\V2_dpskw\\qwen_integration.py\", line 405, in resolved_layers\n raise ValueError(f\"layer_indices must be inside [0, {num_hidden_layers})\")\nValueError: layer_indices must be inside [0, 2)\n"
},
{
"model": "hf-internal-testing/tiny-random-Qwen2ForCausalLM",
"full_stack": false,
"stages": {
"1_config": "FAIL OSError: hf-internal-testing/tiny-random-Qwen2ForCausalLM is not a local folder and is not a valid model identifier listed on 'https://huggingface.co/models'"
},
"patches": []
},
{
"model": "hf-internal-testing/tiny-random-OlmoeForCausalLM",
"full_stack": false,
"stages": {
"1_config": "needs patch: config.text_config",
"2_load_causal_lm": "ok",
"3_container": "needs patch: repo wants model.language_model.layers, found model.layers",
"4_convention": "MISMATCH: second positional is 'attention_mask', adapter forwards position_embeddings positionally",
"5_surgery": "FAIL ValueError: layer_indices must be inside [0, 2)"
},
"patches": [
"config.text_config = config",
"model.model.language_model = model.model"
],
"architecture": "OlmoeForCausalLM",
"model_type": "olmoe",
"hidden_size": 64,
"num_layers": 2,
"vocab_size": 50304,
"text_config_present": false,
"backbone_class": "OlmoeForCausalLM",
"container_path": "model.layers",
"repo_layer_path_ok": false,
"layer_forward_params": [
"hidden_states",
"attention_mask",
"position_ids",
"past_key_values",
"use_cache",
"position_embeddings",
"kwargs"
],
"second_positional_is_position_embeddings": false,
"traceback": "Traceback (most recent call last):\n File \"H:\\Memory\\V2_dpskw\\probe_nm2_portability.py\", line 219, in probe\n layer_indices = memory_config.resolved_layers(num_layers)\n ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^\n File \"H:\\Memory\\V2_dpskw\\qwen_integration.py\", line 405, in resolved_layers\n raise ValueError(f\"layer_indices must be inside [0, {num_hidden_layers})\")\nValueError: layer_indices must be inside [0, 2)\n"
}
]