基于第一性原理 + Anthropic 2026 J-space 论文实现的具身智慧系统。 核心架构: - ODE 动力系统 + 并行专家 + J-space 工作空间广播 - 12 个异构专家(视觉/屏幕/听觉/语言/鼠标/跨模态) - workspace 256 维 + LayerNorm + RK4 积分 自主心智(最重要的能力): - 好奇心驱动探索(内在奖励 + 世界模型) - 跨会话状态持久化(海洋不蒸发) - 自我模型(知道自己会什么不会什么) - 元学习(自适应学习率 + 策略选择) 具身 Agent(完整神经系统): - 感知层:摄像头 + 麦克风 + 屏幕 + 键盘 + 鼠标 - 大脑皮层(workspace)+ 小脑(运动控制)+ 中枢神经(门控) - 海马体(情景记忆)+ 基底神经节(动作选择) - 执行器:鼠标控制 + 键盘输出 + 音频播放 + 屏幕绘制 多模态支持: - 原生图像/音频/视频/文本/键盘/鼠标 6 种模态 - 跨平台(macOS/Windows/Linux) 外挂模块系统: - 可热插拔的外部能力(小模型/知识库/工具) - 核心心智不依赖外挂,断开后继续工作 守护进程: - 用户主动 start/stop(不自启) - 后台静默运行,持续感知学习 - 状态自动保存,跨会话继续 J-lens 可解释性: - 观测模型内部每个 ODE 子步的想法 - Directed Modulation 验证 workspace 因果作用 - Selectivity 验证(ablate workspace) 小模型蒸馏: - 接 GPT-2/Qwen 等迁移理解能力 - 蒸馏完成后小模型可断开 验证结果: - 连续序列:JSpace 胜 Flat 39.7% - 语言进化:loss 3.95→2.40 - workspace ||w||:v1 0.05 → v2 16.0 - 实时五通道感知 + 具身闭环运行
390 lines
15 KiB
Python
390 lines
15 KiB
Python
"""
|
||
自主心智(AutonomousMind)—— 智慧最重要的能力
|
||
|
||
四个核心能力:
|
||
1. CuriosityDrive: 好奇心驱动的主动探索(内在奖励)
|
||
2. PersistentState: 跨会话状态持久化(海洋不蒸发)
|
||
3. SelfModel: 自我模型(知道自己会什么不会什么)
|
||
4. MetaLearner: 元学习(学会如何学习)
|
||
|
||
合起来 = AutonomousMind,永不停止的自主进化。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import torch
|
||
import torch.nn as nn
|
||
import torch.nn.functional as F
|
||
import numpy as np
|
||
import json
|
||
import time
|
||
from pathlib import Path
|
||
from typing import Optional
|
||
from dataclasses import dataclass
|
||
from collections import deque
|
||
|
||
|
||
class CuriosityDrive(nn.Module):
|
||
"""好奇心驱动——基于预测误差的内在奖励
|
||
|
||
模型有世界模型预测下一状态,预测误差=好奇心=内在奖励。
|
||
模型被驱动去探索"预测不准"的区域。学会后好奇心降低,转向新区域。
|
||
"""
|
||
|
||
def __init__(self, workspace_dim: int, action_dim: int = 5, hidden_dim: int = 64):
|
||
super().__init__()
|
||
self.workspace_dim = workspace_dim
|
||
self.world_model = nn.Sequential(
|
||
nn.Linear(workspace_dim + action_dim, hidden_dim),
|
||
nn.ReLU(),
|
||
nn.Linear(hidden_dim, hidden_dim),
|
||
nn.ReLU(),
|
||
nn.Linear(hidden_dim, workspace_dim),
|
||
)
|
||
self.state_history: deque = deque(maxlen=500)
|
||
self.prediction_error_ema = 0.1
|
||
|
||
def predict_next(self, w, action):
|
||
return self.world_model(torch.cat([w, action], dim=-1))
|
||
|
||
def compute_curiosity(self, w_current, action, w_next):
|
||
with torch.no_grad():
|
||
w_pred = self.predict_next(w_current, action)
|
||
pred_error = F.mse_loss(w_pred, w_next).item()
|
||
w_np = w_current[0].cpu().numpy()
|
||
novelty = self._compute_novelty(w_np)
|
||
progress = max(0, pred_error - self.prediction_error_ema * 0.9)
|
||
self.prediction_error_ema = 0.95 * self.prediction_error_ema + 0.05 * pred_error
|
||
curiosity = progress + 0.3 * novelty
|
||
self.state_history.append(w_np.copy())
|
||
return curiosity
|
||
|
||
def _compute_novelty(self, w):
|
||
if len(self.state_history) < 5:
|
||
return 1.0
|
||
history = list(self.state_history)[-100:]
|
||
distances = [np.linalg.norm(w - h) for h in history]
|
||
return float(min(1.0, min(distances) / 2.0))
|
||
|
||
def train_world_model(self, w_current, action, w_next):
|
||
w_pred = self.predict_next(w_current, action.detach())
|
||
return F.mse_loss(w_pred, w_next.detach())
|
||
|
||
|
||
class PersistentState:
|
||
"""跨会话状态持久化——让海洋不蒸发
|
||
|
||
保存 workspace + 专家状态 + 海马体 + 基底神经节 + 好奇心历史 + 自我模型。
|
||
"""
|
||
|
||
def __init__(self, save_dir: Path):
|
||
self.save_dir = Path(save_dir)
|
||
self.save_dir.mkdir(parents=True, exist_ok=True)
|
||
self.state_file = self.save_dir / 'mind_state.json'
|
||
self.tensors_file = self.save_dir / 'mind_tensors.npz'
|
||
|
||
def save(self, state: dict):
|
||
tensors = {}
|
||
if 'w' in state:
|
||
tensors['w'] = state['w'].cpu().numpy()
|
||
if 'm' in state:
|
||
for i, m in enumerate(state['m']):
|
||
if m is not None:
|
||
tensors[f'm_{i}'] = m.cpu().numpy()
|
||
if 'basal_ganglia' in state and state['basal_ganglia'] is not None:
|
||
tensors['bg_weights'] = state['basal_ganglia']
|
||
if 'curiosity_history' in state:
|
||
tensors['curiosity_history'] = np.array(state['curiosity_history'])
|
||
if tensors:
|
||
np.savez(self.tensors_file, **tensors)
|
||
|
||
json_state = {
|
||
'step_count': state.get('step_count', 0),
|
||
'total_runtime': state.get('total_runtime', 0.0),
|
||
'self_model': state.get('self_model', {}),
|
||
'saved_at': time.time(),
|
||
}
|
||
self.state_file.write_text(json.dumps(json_state, indent=2, default=str))
|
||
|
||
def load(self) -> Optional[dict]:
|
||
if not self.state_file.exists():
|
||
return None
|
||
result = {}
|
||
if self.tensors_file.exists():
|
||
data = np.load(self.tensors_file, allow_pickle=True)
|
||
if 'w' in data:
|
||
result['w'] = torch.tensor(data['w'])
|
||
ms = {}
|
||
for key in data.files:
|
||
if key.startswith('m_'):
|
||
idx = int(key.split('_')[1])
|
||
ms[idx] = torch.tensor(data[key])
|
||
if ms:
|
||
result['m'] = [ms[i] for i in sorted(ms.keys())]
|
||
if 'bg_weights' in data:
|
||
result['basal_ganglia'] = data['bg_weights']
|
||
if 'curiosity_history' in data:
|
||
result['curiosity_history'] = data['curiosity_history'].tolist()
|
||
json_state = json.loads(self.state_file.read_text())
|
||
result.update(json_state)
|
||
return result
|
||
|
||
|
||
class SelfModel:
|
||
"""自我模型——知道自己会什么、不会什么
|
||
|
||
对每个能力域维护置信度(0-1),通过历史成功率更新。
|
||
不知道时主动学习(好奇心驱动)。
|
||
"""
|
||
|
||
def __init__(self, capabilities=None):
|
||
if capabilities is None:
|
||
capabilities = ['visual', 'audio', 'text', 'motor_mouse',
|
||
'motor_keyboard', 'memory', 'prediction']
|
||
self.capabilities = capabilities
|
||
self.confidence = {c: 0.0 for c in capabilities}
|
||
self.attempts = {c: 0 for c in capabilities}
|
||
self.recent_results = {c: deque(maxlen=20) for c in capabilities}
|
||
|
||
def record_attempt(self, capability, success):
|
||
if capability not in self.confidence:
|
||
return
|
||
self.attempts[capability] += 1
|
||
self.recent_results[capability].append(success)
|
||
recent = list(self.recent_results[capability])
|
||
if recent:
|
||
weights = np.linspace(0.5, 1.0, len(recent))
|
||
self.confidence[capability] = float(np.average(recent, weights=weights))
|
||
|
||
def get_weakness(self):
|
||
return min(self.confidence, key=self.confidence.get)
|
||
|
||
def get_strength(self):
|
||
return max(self.confidence, key=self.confidence.get)
|
||
|
||
def knows(self, capability, threshold=0.5):
|
||
return self.confidence.get(capability, 0.0) > threshold
|
||
|
||
def summary(self):
|
||
return {
|
||
'capabilities': dict(self.confidence),
|
||
'attempts': dict(self.attempts),
|
||
'strength': self.get_strength(),
|
||
'weakness': self.get_weakness(),
|
||
}
|
||
|
||
|
||
class MetaLearner:
|
||
"""元学习——学会如何学习
|
||
|
||
每个能力域有自适应学习率:进步快→加快,停滞→减慢。
|
||
记录什么学习策略有效。
|
||
"""
|
||
|
||
def __init__(self, capabilities=None):
|
||
if capabilities is None:
|
||
capabilities = ['visual', 'audio', 'text', 'motor_mouse',
|
||
'motor_keyboard', 'memory', 'prediction']
|
||
self.learning_rates = {c: 1e-3 for c in capabilities}
|
||
self.loss_history = {c: deque(maxlen=20) for c in capabilities}
|
||
self.strategy_scores = {
|
||
'predict_next': 0.5, 'replay': 0.5,
|
||
'explore': 0.5, 'imitate': 0.5,
|
||
}
|
||
|
||
def get_lr(self, capability):
|
||
return self.learning_rates.get(capability, 1e-3)
|
||
|
||
def record_loss(self, capability, loss):
|
||
if capability not in self.loss_history:
|
||
return
|
||
self.loss_history[capability].append(loss)
|
||
history = list(self.loss_history[capability])
|
||
if len(history) < 5:
|
||
return
|
||
recent_avg = np.mean(history[-5:])
|
||
old_avg = np.mean(history[-10:-5]) if len(history) >= 10 else recent_avg
|
||
improvement = (old_avg - recent_avg) / max(old_avg, 1e-8)
|
||
lr = self.learning_rates[capability]
|
||
if improvement > 0.05:
|
||
lr *= 1.1
|
||
elif improvement < 0.01:
|
||
lr *= 0.9
|
||
self.learning_rates[capability] = max(1e-5, min(1e-2, lr))
|
||
|
||
def best_strategy(self):
|
||
return max(self.strategy_scores, key=self.strategy_scores.get)
|
||
|
||
def reward_strategy(self, strategy, reward):
|
||
if strategy in self.strategy_scores:
|
||
self.strategy_scores[strategy] = (
|
||
0.9 * self.strategy_scores[strategy] + 0.1 * reward
|
||
)
|
||
|
||
|
||
class AutonomousMind:
|
||
"""自主心智——永不停止的自主进化
|
||
|
||
整合好奇心 + 持久化 + 自我模型 + 元学习。
|
||
核心循环:感知→自我评估→好奇探索→行动→观察→学习→存盘
|
||
"""
|
||
|
||
def __init__(self, agent, save_dir='outputs/mind', device='cpu'):
|
||
self.agent = agent
|
||
self.device = device
|
||
self.config = agent.config
|
||
|
||
self.curiosity = CuriosityDrive(
|
||
workspace_dim=self.config.workspace_dim, action_dim=5,
|
||
).to(device)
|
||
self.persistence = PersistentState(Path(save_dir))
|
||
self.self_model = SelfModel()
|
||
self.meta_learner = MetaLearner()
|
||
|
||
self.step_count = 0
|
||
self.total_runtime = 0.0
|
||
self.start_time = time.time()
|
||
self.running = False
|
||
self.curiosity_history = deque(maxlen=1000)
|
||
self.curiosity_optimizer = torch.optim.Adam(
|
||
self.curiosity.parameters(), lr=1e-3
|
||
)
|
||
|
||
self._load_state()
|
||
|
||
def _load_state(self):
|
||
state = self.persistence.load()
|
||
if state is None:
|
||
print(" [心智] 全新启动")
|
||
return
|
||
print(f" [心智] 恢复状态: step={state.get('step_count', 0)}")
|
||
if 'w' in state:
|
||
self.agent.state['w'] = state['w'].to(self.device)
|
||
if 'm' in state:
|
||
for i, m in enumerate(state['m']):
|
||
if m is not None and i < len(self.agent.state['m']):
|
||
self.agent.state['m'][i] = m.to(self.device)
|
||
if 'basal_ganglia' in state and hasattr(self.agent, 'basal_ganglia'):
|
||
self.agent.basal_ganglia.action_weights = state['basal_ganglia']
|
||
self.step_count = state.get('step_count', 0)
|
||
self.total_runtime = state.get('total_runtime', 0.0)
|
||
|
||
def save_state(self):
|
||
state = {
|
||
'w': self.agent.state['w'],
|
||
'm': self.agent.state['m'],
|
||
'basal_ganglia': getattr(self.agent.basal_ganglia, 'action_weights', None)
|
||
if hasattr(self.agent, 'basal_ganglia') else None,
|
||
'curiosity_history': list(self.curiosity.state_history),
|
||
'step_count': self.step_count,
|
||
'total_runtime': self.total_runtime + (time.time() - self.start_time),
|
||
'self_model': {'confidence': self.self_model.confidence},
|
||
}
|
||
self.persistence.save(state)
|
||
|
||
def step(self) -> dict:
|
||
# 1. 感知 + 思考
|
||
sensory_data = self.agent.perceive()
|
||
w_before = self.agent.state['w'].clone()
|
||
weakness = self.self_model.get_weakness()
|
||
strength = self.self_model.get_strength()
|
||
|
||
# 2. 行动
|
||
action_info = self.agent.decide_and_act(w_before, sensory_data.get('modality', 'idle'))
|
||
w_after, modality = self.agent.think(sensory_data)
|
||
|
||
# 3. 好奇心
|
||
action_tensor = torch.tensor(action_info['action_params'],
|
||
dtype=torch.float32).unsqueeze(0).to(self.device)
|
||
curiosity_reward = self.curiosity.compute_curiosity(w_before, action_tensor, w_after)
|
||
self.curiosity_history.append(curiosity_reward)
|
||
|
||
# 4. 训练世界模型
|
||
world_loss = self.curiosity.train_world_model(w_before, action_tensor, w_after)
|
||
self.curiosity_optimizer.zero_grad()
|
||
world_loss.backward()
|
||
self.curiosity_optimizer.step()
|
||
|
||
# 5. 评估成功度
|
||
w_stability = 1.0 - min(1.0, abs(w_after.norm().item() - w_before.norm().item()))
|
||
success = (0.3 * float(action_info['executed']) +
|
||
0.4 * min(1.0, curiosity_reward) + 0.3 * w_stability)
|
||
|
||
# 6. 更新自我模型
|
||
cap_map = {'image': 'visual', 'screen': 'visual', 'audio': 'audio',
|
||
'text': 'text', 'keyboard': 'text', 'mouse': 'motor_mouse', 'idle': 'prediction'}
|
||
cap = cap_map.get(modality, 'prediction')
|
||
self.self_model.record_attempt(cap, success)
|
||
self.self_model.record_attempt('prediction', 1.0 - min(1.0, world_loss.item()))
|
||
|
||
# 7. 元学习
|
||
self.meta_learner.record_loss(cap, world_loss.item())
|
||
if curiosity_reward > 0.3:
|
||
self.meta_learner.reward_strategy('explore', curiosity_reward)
|
||
|
||
# 8. 记忆 + 基底神经节
|
||
self.agent.remember(w_after, {
|
||
'modality': modality, 'curiosity': curiosity_reward,
|
||
'success': success, 'step': self.step_count,
|
||
})
|
||
self.agent.learn(w_after, np.array(action_info['action_params']), reward=curiosity_reward)
|
||
|
||
self.step_count += 1
|
||
return {
|
||
'step': self.step_count, 'modality': modality,
|
||
'w_norm': w_after.norm().item(), 'curiosity': curiosity_reward,
|
||
'world_loss': world_loss.item(), 'success': success,
|
||
'weakness': weakness, 'strength': strength,
|
||
'self_confidence': dict(self.self_model.confidence),
|
||
'best_strategy': self.meta_learner.best_strategy(),
|
||
'memory_count': self.agent.hippocampus.size() if self.agent.hippocampus else 0,
|
||
}
|
||
|
||
def run(self, n_steps=100, interval=0.2, save_every=50, on_step=None):
|
||
self.running = True
|
||
self.start_time = time.time()
|
||
self.agent.senses.start()
|
||
print(f"\n自主心智启动 | 总步数: {self.step_count} | 保存间隔: {save_every}")
|
||
print("=" * 60)
|
||
|
||
log = []
|
||
try:
|
||
for _ in range(n_steps):
|
||
if not self.running:
|
||
break
|
||
info = self.step()
|
||
log.append(info)
|
||
if on_step:
|
||
on_step(info)
|
||
elif info['step'] % 10 == 0:
|
||
print(f" step {info['step']:4d} | mod {info['modality']:8s} | "
|
||
f"||w|| {info['w_norm']:.3f} | curio {info['curiosity']:.3f} | "
|
||
f"success {info['success']:.2f} | weak={info['weakness']} | "
|
||
f"mem {info['memory_count']}")
|
||
if info['step'] % save_every == 0:
|
||
self.save_state()
|
||
time.sleep(interval)
|
||
except KeyboardInterrupt:
|
||
print("\n用户中断")
|
||
finally:
|
||
self.running = False
|
||
self.agent.senses.stop()
|
||
if hasattr(self.agent, 'audio_actuator'):
|
||
self.agent.audio_actuator.stop()
|
||
self.save_state()
|
||
self.total_runtime += time.time() - self.start_time
|
||
return log
|
||
|
||
def introspect(self) -> str:
|
||
"""内省——自我报告"""
|
||
sm = self.self_model.summary()
|
||
avg_curio = np.mean(list(self.curiosity_history)) if self.curiosity_history else 0
|
||
report = f"=== 自主心智内省 ===\n步数: {self.step_count}\n运行: {self.total_runtime:.0f}s\n"
|
||
report += f"记忆: {self.agent.hippocampus.size() if self.agent.hippocampus else 0}\n"
|
||
report += f"平均好奇心: {avg_curio:.3f}\n\n自我认知:\n"
|
||
for cap, conf in sm['capabilities'].items():
|
||
bar = '█' * int(conf * 20)
|
||
report += f" {cap:15s}: {conf:.2f} {bar}\n"
|
||
report += f"\n最强: {sm['strength']}\n最弱: {sm['weakness']}\n"
|
||
report += f"最佳策略: {self.meta_learner.best_strategy()}\n"
|
||
return report
|