feat: JspaceAI 自主智慧架构

基于第一性原理 + Anthropic 2026 J-space 论文实现的具身智慧系统。

核心架构:
- ODE 动力系统 + 并行专家 + J-space 工作空间广播
- 12 个异构专家(视觉/屏幕/听觉/语言/鼠标/跨模态)
- workspace 256 维 + LayerNorm + RK4 积分

自主心智(最重要的能力):
- 好奇心驱动探索(内在奖励 + 世界模型)
- 跨会话状态持久化(海洋不蒸发)
- 自我模型(知道自己会什么不会什么)
- 元学习(自适应学习率 + 策略选择)

具身 Agent(完整神经系统):
- 感知层:摄像头 + 麦克风 + 屏幕 + 键盘 + 鼠标
- 大脑皮层(workspace)+ 小脑(运动控制)+ 中枢神经(门控)
- 海马体(情景记忆)+ 基底神经节(动作选择)
- 执行器:鼠标控制 + 键盘输出 + 音频播放 + 屏幕绘制

多模态支持:
- 原生图像/音频/视频/文本/键盘/鼠标 6 种模态
- 跨平台(macOS/Windows/Linux)

外挂模块系统:
- 可热插拔的外部能力(小模型/知识库/工具)
- 核心心智不依赖外挂,断开后继续工作

守护进程:
- 用户主动 start/stop(不自启)
- 后台静默运行,持续感知学习
- 状态自动保存,跨会话继续

J-lens 可解释性:
- 观测模型内部每个 ODE 子步的想法
- Directed Modulation 验证 workspace 因果作用
- Selectivity 验证(ablate workspace)

小模型蒸馏:
- 接 GPT-2/Qwen 等迁移理解能力
- 蒸馏完成后小模型可断开

验证结果:
- 连续序列:JSpace 胜 Flat 39.7%
- 语言进化:loss 3.95→2.40
- workspace ||w||:v1 0.05 → v2 16.0
- 实时五通道感知 + 具身闭环运行
This commit is contained in:
2026-07-07 09:19:31 +08:00
commit e782ef4db1
30 changed files with 7446 additions and 0 deletions

389
jspaceai/autonomous.py Normal file
View File

@@ -0,0 +1,389 @@
"""
自主心智AutonomousMind—— 智慧最重要的能力
四个核心能力:
1. CuriosityDrive: 好奇心驱动的主动探索(内在奖励)
2. PersistentState: 跨会话状态持久化(海洋不蒸发)
3. SelfModel: 自我模型(知道自己会什么不会什么)
4. MetaLearner: 元学习(学会如何学习)
合起来 = AutonomousMind永不停止的自主进化。
"""
from __future__ import annotations
import torch
import torch.nn as nn
import torch.nn.functional as F
import numpy as np
import json
import time
from pathlib import Path
from typing import Optional
from dataclasses import dataclass
from collections import deque
class CuriosityDrive(nn.Module):
"""好奇心驱动——基于预测误差的内在奖励
模型有世界模型预测下一状态,预测误差=好奇心=内在奖励。
模型被驱动去探索"预测不准"的区域。学会后好奇心降低,转向新区域。
"""
def __init__(self, workspace_dim: int, action_dim: int = 5, hidden_dim: int = 64):
super().__init__()
self.workspace_dim = workspace_dim
self.world_model = nn.Sequential(
nn.Linear(workspace_dim + action_dim, hidden_dim),
nn.ReLU(),
nn.Linear(hidden_dim, hidden_dim),
nn.ReLU(),
nn.Linear(hidden_dim, workspace_dim),
)
self.state_history: deque = deque(maxlen=500)
self.prediction_error_ema = 0.1
def predict_next(self, w, action):
return self.world_model(torch.cat([w, action], dim=-1))
def compute_curiosity(self, w_current, action, w_next):
with torch.no_grad():
w_pred = self.predict_next(w_current, action)
pred_error = F.mse_loss(w_pred, w_next).item()
w_np = w_current[0].cpu().numpy()
novelty = self._compute_novelty(w_np)
progress = max(0, pred_error - self.prediction_error_ema * 0.9)
self.prediction_error_ema = 0.95 * self.prediction_error_ema + 0.05 * pred_error
curiosity = progress + 0.3 * novelty
self.state_history.append(w_np.copy())
return curiosity
def _compute_novelty(self, w):
if len(self.state_history) < 5:
return 1.0
history = list(self.state_history)[-100:]
distances = [np.linalg.norm(w - h) for h in history]
return float(min(1.0, min(distances) / 2.0))
def train_world_model(self, w_current, action, w_next):
w_pred = self.predict_next(w_current, action.detach())
return F.mse_loss(w_pred, w_next.detach())
class PersistentState:
"""跨会话状态持久化——让海洋不蒸发
保存 workspace + 专家状态 + 海马体 + 基底神经节 + 好奇心历史 + 自我模型。
"""
def __init__(self, save_dir: Path):
self.save_dir = Path(save_dir)
self.save_dir.mkdir(parents=True, exist_ok=True)
self.state_file = self.save_dir / 'mind_state.json'
self.tensors_file = self.save_dir / 'mind_tensors.npz'
def save(self, state: dict):
tensors = {}
if 'w' in state:
tensors['w'] = state['w'].cpu().numpy()
if 'm' in state:
for i, m in enumerate(state['m']):
if m is not None:
tensors[f'm_{i}'] = m.cpu().numpy()
if 'basal_ganglia' in state and state['basal_ganglia'] is not None:
tensors['bg_weights'] = state['basal_ganglia']
if 'curiosity_history' in state:
tensors['curiosity_history'] = np.array(state['curiosity_history'])
if tensors:
np.savez(self.tensors_file, **tensors)
json_state = {
'step_count': state.get('step_count', 0),
'total_runtime': state.get('total_runtime', 0.0),
'self_model': state.get('self_model', {}),
'saved_at': time.time(),
}
self.state_file.write_text(json.dumps(json_state, indent=2, default=str))
def load(self) -> Optional[dict]:
if not self.state_file.exists():
return None
result = {}
if self.tensors_file.exists():
data = np.load(self.tensors_file, allow_pickle=True)
if 'w' in data:
result['w'] = torch.tensor(data['w'])
ms = {}
for key in data.files:
if key.startswith('m_'):
idx = int(key.split('_')[1])
ms[idx] = torch.tensor(data[key])
if ms:
result['m'] = [ms[i] for i in sorted(ms.keys())]
if 'bg_weights' in data:
result['basal_ganglia'] = data['bg_weights']
if 'curiosity_history' in data:
result['curiosity_history'] = data['curiosity_history'].tolist()
json_state = json.loads(self.state_file.read_text())
result.update(json_state)
return result
class SelfModel:
"""自我模型——知道自己会什么、不会什么
对每个能力域维护置信度0-1通过历史成功率更新。
不知道时主动学习(好奇心驱动)。
"""
def __init__(self, capabilities=None):
if capabilities is None:
capabilities = ['visual', 'audio', 'text', 'motor_mouse',
'motor_keyboard', 'memory', 'prediction']
self.capabilities = capabilities
self.confidence = {c: 0.0 for c in capabilities}
self.attempts = {c: 0 for c in capabilities}
self.recent_results = {c: deque(maxlen=20) for c in capabilities}
def record_attempt(self, capability, success):
if capability not in self.confidence:
return
self.attempts[capability] += 1
self.recent_results[capability].append(success)
recent = list(self.recent_results[capability])
if recent:
weights = np.linspace(0.5, 1.0, len(recent))
self.confidence[capability] = float(np.average(recent, weights=weights))
def get_weakness(self):
return min(self.confidence, key=self.confidence.get)
def get_strength(self):
return max(self.confidence, key=self.confidence.get)
def knows(self, capability, threshold=0.5):
return self.confidence.get(capability, 0.0) > threshold
def summary(self):
return {
'capabilities': dict(self.confidence),
'attempts': dict(self.attempts),
'strength': self.get_strength(),
'weakness': self.get_weakness(),
}
class MetaLearner:
"""元学习——学会如何学习
每个能力域有自适应学习率:进步快→加快,停滞→减慢。
记录什么学习策略有效。
"""
def __init__(self, capabilities=None):
if capabilities is None:
capabilities = ['visual', 'audio', 'text', 'motor_mouse',
'motor_keyboard', 'memory', 'prediction']
self.learning_rates = {c: 1e-3 for c in capabilities}
self.loss_history = {c: deque(maxlen=20) for c in capabilities}
self.strategy_scores = {
'predict_next': 0.5, 'replay': 0.5,
'explore': 0.5, 'imitate': 0.5,
}
def get_lr(self, capability):
return self.learning_rates.get(capability, 1e-3)
def record_loss(self, capability, loss):
if capability not in self.loss_history:
return
self.loss_history[capability].append(loss)
history = list(self.loss_history[capability])
if len(history) < 5:
return
recent_avg = np.mean(history[-5:])
old_avg = np.mean(history[-10:-5]) if len(history) >= 10 else recent_avg
improvement = (old_avg - recent_avg) / max(old_avg, 1e-8)
lr = self.learning_rates[capability]
if improvement > 0.05:
lr *= 1.1
elif improvement < 0.01:
lr *= 0.9
self.learning_rates[capability] = max(1e-5, min(1e-2, lr))
def best_strategy(self):
return max(self.strategy_scores, key=self.strategy_scores.get)
def reward_strategy(self, strategy, reward):
if strategy in self.strategy_scores:
self.strategy_scores[strategy] = (
0.9 * self.strategy_scores[strategy] + 0.1 * reward
)
class AutonomousMind:
"""自主心智——永不停止的自主进化
整合好奇心 + 持久化 + 自我模型 + 元学习。
核心循环:感知→自我评估→好奇探索→行动→观察→学习→存盘
"""
def __init__(self, agent, save_dir='outputs/mind', device='cpu'):
self.agent = agent
self.device = device
self.config = agent.config
self.curiosity = CuriosityDrive(
workspace_dim=self.config.workspace_dim, action_dim=5,
).to(device)
self.persistence = PersistentState(Path(save_dir))
self.self_model = SelfModel()
self.meta_learner = MetaLearner()
self.step_count = 0
self.total_runtime = 0.0
self.start_time = time.time()
self.running = False
self.curiosity_history = deque(maxlen=1000)
self.curiosity_optimizer = torch.optim.Adam(
self.curiosity.parameters(), lr=1e-3
)
self._load_state()
def _load_state(self):
state = self.persistence.load()
if state is None:
print(" [心智] 全新启动")
return
print(f" [心智] 恢复状态: step={state.get('step_count', 0)}")
if 'w' in state:
self.agent.state['w'] = state['w'].to(self.device)
if 'm' in state:
for i, m in enumerate(state['m']):
if m is not None and i < len(self.agent.state['m']):
self.agent.state['m'][i] = m.to(self.device)
if 'basal_ganglia' in state and hasattr(self.agent, 'basal_ganglia'):
self.agent.basal_ganglia.action_weights = state['basal_ganglia']
self.step_count = state.get('step_count', 0)
self.total_runtime = state.get('total_runtime', 0.0)
def save_state(self):
state = {
'w': self.agent.state['w'],
'm': self.agent.state['m'],
'basal_ganglia': getattr(self.agent.basal_ganglia, 'action_weights', None)
if hasattr(self.agent, 'basal_ganglia') else None,
'curiosity_history': list(self.curiosity.state_history),
'step_count': self.step_count,
'total_runtime': self.total_runtime + (time.time() - self.start_time),
'self_model': {'confidence': self.self_model.confidence},
}
self.persistence.save(state)
def step(self) -> dict:
# 1. 感知 + 思考
sensory_data = self.agent.perceive()
w_before = self.agent.state['w'].clone()
weakness = self.self_model.get_weakness()
strength = self.self_model.get_strength()
# 2. 行动
action_info = self.agent.decide_and_act(w_before, sensory_data.get('modality', 'idle'))
w_after, modality = self.agent.think(sensory_data)
# 3. 好奇心
action_tensor = torch.tensor(action_info['action_params'],
dtype=torch.float32).unsqueeze(0).to(self.device)
curiosity_reward = self.curiosity.compute_curiosity(w_before, action_tensor, w_after)
self.curiosity_history.append(curiosity_reward)
# 4. 训练世界模型
world_loss = self.curiosity.train_world_model(w_before, action_tensor, w_after)
self.curiosity_optimizer.zero_grad()
world_loss.backward()
self.curiosity_optimizer.step()
# 5. 评估成功度
w_stability = 1.0 - min(1.0, abs(w_after.norm().item() - w_before.norm().item()))
success = (0.3 * float(action_info['executed']) +
0.4 * min(1.0, curiosity_reward) + 0.3 * w_stability)
# 6. 更新自我模型
cap_map = {'image': 'visual', 'screen': 'visual', 'audio': 'audio',
'text': 'text', 'keyboard': 'text', 'mouse': 'motor_mouse', 'idle': 'prediction'}
cap = cap_map.get(modality, 'prediction')
self.self_model.record_attempt(cap, success)
self.self_model.record_attempt('prediction', 1.0 - min(1.0, world_loss.item()))
# 7. 元学习
self.meta_learner.record_loss(cap, world_loss.item())
if curiosity_reward > 0.3:
self.meta_learner.reward_strategy('explore', curiosity_reward)
# 8. 记忆 + 基底神经节
self.agent.remember(w_after, {
'modality': modality, 'curiosity': curiosity_reward,
'success': success, 'step': self.step_count,
})
self.agent.learn(w_after, np.array(action_info['action_params']), reward=curiosity_reward)
self.step_count += 1
return {
'step': self.step_count, 'modality': modality,
'w_norm': w_after.norm().item(), 'curiosity': curiosity_reward,
'world_loss': world_loss.item(), 'success': success,
'weakness': weakness, 'strength': strength,
'self_confidence': dict(self.self_model.confidence),
'best_strategy': self.meta_learner.best_strategy(),
'memory_count': self.agent.hippocampus.size() if self.agent.hippocampus else 0,
}
def run(self, n_steps=100, interval=0.2, save_every=50, on_step=None):
self.running = True
self.start_time = time.time()
self.agent.senses.start()
print(f"\n自主心智启动 | 总步数: {self.step_count} | 保存间隔: {save_every}")
print("=" * 60)
log = []
try:
for _ in range(n_steps):
if not self.running:
break
info = self.step()
log.append(info)
if on_step:
on_step(info)
elif info['step'] % 10 == 0:
print(f" step {info['step']:4d} | mod {info['modality']:8s} | "
f"||w|| {info['w_norm']:.3f} | curio {info['curiosity']:.3f} | "
f"success {info['success']:.2f} | weak={info['weakness']} | "
f"mem {info['memory_count']}")
if info['step'] % save_every == 0:
self.save_state()
time.sleep(interval)
except KeyboardInterrupt:
print("\n用户中断")
finally:
self.running = False
self.agent.senses.stop()
if hasattr(self.agent, 'audio_actuator'):
self.agent.audio_actuator.stop()
self.save_state()
self.total_runtime += time.time() - self.start_time
return log
def introspect(self) -> str:
"""内省——自我报告"""
sm = self.self_model.summary()
avg_curio = np.mean(list(self.curiosity_history)) if self.curiosity_history else 0
report = f"=== 自主心智内省 ===\n步数: {self.step_count}\n运行: {self.total_runtime:.0f}s\n"
report += f"记忆: {self.agent.hippocampus.size() if self.agent.hippocampus else 0}\n"
report += f"平均好奇心: {avg_curio:.3f}\n\n自我认知:\n"
for cap, conf in sm['capabilities'].items():
bar = '' * int(conf * 20)
report += f" {cap:15s}: {conf:.2f} {bar}\n"
report += f"\n最强: {sm['strength']}\n最弱: {sm['weakness']}\n"
report += f"最佳策略: {self.meta_learner.best_strategy()}\n"
return report