diff --git a/agent_loop.py b/agent_loop.py new file mode 100644 index 0000000..0e66e39 --- /dev/null +++ b/agent_loop.py @@ -0,0 +1,163 @@ +import socket +import json +import time +from openai import OpenAI + +# 初始化 DeepSeek 客户端 +client = OpenAI( + api_key="sk-420190f448fe41158c4e2ccff90e35ce", + base_url="https://api.deepseek.com" +) +MODEL_NAME = "deepseek-chat" + +def fetch_and_translate_tools(): + """向 CAD 基座发送 tools/list,并转换为大模型认识的格式""" + try: + cad_socket = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + cad_socket.connect(("127.0.0.1", 8080)) + + # 1. 向 MCP Server 发送标准的 tools/list 请求 + req = { + "jsonrpc": "2.0", + "id": 100, + "method": "tools/list" + } + cad_socket.sendall(json.dumps(req).encode('utf-8')) + + response_data = cad_socket.recv(8192) + cad_socket.close() + + mcp_response = json.loads(response_data.decode('utf-8')) + + if "result" not in mcp_response or "tools" not in mcp_response["result"]: + print("[Error] 无法从 CAD 获取工具列表") + return [] + + llm_tools = [] + # 2. 遍历 CAD 返回的工具,将其翻译为 DeepSeek/OpenAI 格式 + for mcp_tool in mcp_response["result"]["tools"]: + llm_tools.append({ + "type": "function", + "function": { + "name": mcp_tool["name"], + "description": mcp_tool["description"], + # 核心转换:MCP 的 inputSchema 等价于 LLM 的 parameters + "parameters": mcp_tool["inputSchema"] + } + }) + + print(f"[Init] 成功从 CAD 动态加载了 {len(llm_tools)} 个工具!") + return llm_tools + + except Exception as e: + print(f"[Error] 获取工具列表失败: {e}") + return [] + +def call_cad_mcp_server(tool_name, arguments): + """底层 TCP 通信保持不变""" + try: + cad_socket = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + cad_socket.connect(("127.0.0.1", 8080)) + req = { + "jsonrpc": "2.0", "id": 1, "method": "tools/call", + "params": { "name": tool_name, "arguments": arguments } + } + cad_socket.sendall(json.dumps(req).encode('utf-8')) + response_data = b"" + while True: + chunk = cad_socket.recv(8192) + response_data += chunk + if len(chunk) < 8192: break + cad_socket.close() + return json.loads(response_data.decode('utf-8')) + except Exception as e: + print(f"[Error] 连接 CAD 基座失败: {e}") + return None + +def run_agent(user_prompt): + print("=== 🚀 CAD 具身智能 Agent 启动 ===") + + # 动态向 CAD 请求可用工具,彻底解耦! + tools = fetch_and_translate_tools() + + if not tools: + print("没有可用的工具,Agent 退出。") + return + + # 核心心智设定:教它怎么形成视觉闭环 + system_prompt = """你是一个高阶 AutoCAD 视觉检查专家。 +请严格遵循以下工作流: +1. 第一步永远是调用 `get_viewport_screenshot` 观察当前画面。 +2. 仔细评估图像。如果目标物体(如图元、文字)太小导致无法确信细节,请调用 `zoom_window_normalized` 放大该局部区域。 +3. 【关键指令】:每次执行缩放操作(zoom)后,你必须在下一回合再次调用 `get_viewport_screenshot` 获取放大后的新截图! +4. 反复观察和放大,直到你 100% 看清细节,再用自然语言向用户输出最终结论。""" + + messages = [ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_prompt} + ] + + MAX_TURNS = 8 # 熔断机制:最多允许思考 8 个回合,防止无限套娃 + + for turn in range(MAX_TURNS): + print(f"\n--- [第 {turn + 1} 回合] 思考中 ---") + + response = client.chat.completions.create( + model=MODEL_NAME, + messages=messages, + tools=tools + ) + + assistant_message = response.choices[0].message + messages.append(assistant_message) # 记录思维路径 + + # 1. 检查是否需要执行物理动作 + if assistant_message.tool_calls: + for tool_call in assistant_message.tool_calls: + tool_name = tool_call.function.name + args = json.loads(tool_call.function.arguments) + + print(f"[动作决定] ⚡ 调用工具: {tool_name}") + if args: print(f" 参数: {args}") + + # 执行 CAD 通信 + cad_res = call_cad_mcp_server(tool_name, args) + time.sleep(0.5) # 给予 CAD 渲染刷新窗口的时间缓冲 + + # 处理执行结果,并准备发回给 LLM 的观测报告 + tool_result_content = [] + + if cad_res and "result" in cad_res: + for item in cad_res["result"]["content"]: + if item["type"] == "text": + tool_result_content.append({"type": "text", "text": item["text"]}) + elif item["type"] == "image": + base64_img = item["data"] + mime = item.get("mimeType", "image/png") + tool_result_content.append({ + "type": "image_url", + "image_url": {"url": f"data:{mime};base64,{base64_img}"} + }) + print("[观测反馈] 📸 已将最新 CAD 屏幕画面传回视觉中枢。") + else: + tool_result_content.append({"type": "text", "text": "CAD 工具执行失败或无响应。"}) + + # 必须向大模型提交 Tool 返回结果 + messages.append({ + "role": "tool", + "tool_call_id": tool_call.id, + "content": tool_result_content + }) + + # 2. 如果不调工具,说明任务已完成,输出最终自然语言结论 + else: + print("\n==================================") + print(f"🎯 [最终结论]:\n{assistant_message.content}") + print("==================================") + break + + else: + print("\n[警告] 达到最大思考回合数限制,Agent 强行中止。") + +if __name__ == "__main__": + run_agent("请仔细检查当前图纸,告诉我图纸中心那些小圆的内部,是否还包含了更小的同心圆?如果看不清,请务必放大确认。") \ No newline at end of file diff --git a/countOfCircle.py b/countOfCircle.py new file mode 100644 index 0000000..6defc86 --- /dev/null +++ b/countOfCircle.py @@ -0,0 +1,127 @@ +import socket +import json +from openai import OpenAI + +# 1. 初始化 DeepSeek 客户端 +# 注意:务必确保已经执行过 pip install openai +client = OpenAI( + api_key="sk-420190f448fe41158c4e2ccff90e35ce", + base_url="https://api.deepseek.com" # 核心修改:将网关指向 DeepSeek 服务器 +) + +# 核心修改:使用 DeepSeek 的主模型。 +# (注: DeepSeek-V4/V4.1 的视觉支持已集成,具体模型名请以你 DeepSeek 后台显示的可用模型为准) +MODEL_NAME = "deepseek-flash" + +def call_cad_mcp_server(tool_name, arguments): + """封装与 AutoCAD MCP 基座的 TCP 通信""" + try: + cad_socket = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + cad_socket.connect(("127.0.0.1", 8080)) + + req = { + "jsonrpc": "2.0", + "id": 1, + "method": "tools/call", + "params": { + "name": tool_name, + "arguments": arguments + } + } + cad_socket.sendall(json.dumps(req).encode('utf-8')) + + response_data = b"" + while True: + chunk = cad_socket.recv(8192) + response_data += chunk + if len(chunk) < 8192: + break + + cad_socket.close() + return json.loads(response_data.decode('utf-8')) + except Exception as e: + print(f"[Error] 连接 CAD 基座失败: {e}") + return None + +def main(): + print("=== DeepSeek CAD Vision Agent 启动 ===") + + # 2. 定义 MCP 工具 (Tool Calling) + tools = [ + { + "type": "function", + "function": { + "name": "get_viewport_screenshot", + "description": "获取当前 AutoCAD 视口的实时截图。当用户询问图纸上的视觉特征、数量或位置时,必须先调用此工具获取画面。" + } + } + ] + + # 3. 初始化历史 + messages = [ + {"role": "system", "content": "你是一个专业的 AutoCAD 视觉审查助手。必须先使用截图工具观察图纸,再回答用户问题。"}, + {"role": "user", "content": "请看看当前 CAD 屏幕上,我一共画了几个圆?"} + ] + + print("\n[DeepSeek] 正在思考如何完成任务...") + + # === 第一回合:DeepSeek 思考并决定调用工具 === + response = client.chat.completions.create( + model=MODEL_NAME, + messages=messages, + tools=tools + ) + + assistant_message = response.choices[0].message + messages.append(assistant_message) + + # 检查 DeepSeek 是否发起了 Tool Call + if assistant_message.tool_calls: + tool_call = assistant_message.tool_calls[0] + tool_name = tool_call.function.name + + print(f"[DeepSeek] 决定调用 CAD 工具: {tool_name}") + print(f"[CAD] 正在执行截图,请稍候...") + + # === 与 CAD 基座通信 === + cad_res = call_cad_mcp_server(tool_name, {}) + + if cad_res and "result" in cad_res: + content_array = cad_res["result"]["content"] + base64_img = "" + mime_type = "image/png" + for item in content_array: + if item["type"] == "image": + base64_img = item["data"] + mime_type = item.get("mimeType", "image/png") + + print("[CAD] 截图成功!正在将视觉数据传回大模型神经中枢...") + + # 4. 将 Base64 图片按标准多模态格式塞回历史 + messages.append({ + "role": "tool", + "tool_call_id": tool_call.id, + "content": [ + {"type": "text", "text": "这是 AutoCAD 当前界面的实时截图。请根据图片回答用户的问题。"}, + {"type": "image_url", "image_url": {"url": f"data:{mime_type};base64,{base64_img}"}} + ] + }) + + # === 第二回合:DeepSeek “看”图并回答 === + print("[DeepSeek] 正在进行视觉分析...") + final_response = client.chat.completions.create( + model=MODEL_NAME, + messages=messages + ) + + print("\n==================================") + print(f"[DeepSeek 最终回答]:\n{final_response.choices[0].message.content}") + print("==================================") + + else: + print("[Error] CAD 工具执行失败。") + else: + print(f"[DeepSeek 盲猜]: {assistant_message.content}") + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/src/cad_mcp_plugins/mcp_plugins.cpp b/src/cad_mcp_plugins/mcp_plugins.cpp index 31f5b4c..8425c96 100644 --- a/src/cad_mcp_plugins/mcp_plugins.cpp +++ b/src/cad_mcp_plugins/mcp_plugins.cpp @@ -3,74 +3,157 @@ #include #include "tchar.h" #include "rxregsvc.h" +#include +#pragma comment(lib, "gdiplus.lib") #pragma comment(lib, "rxapi.lib") #pragma comment(linker, "/export:acrxGetApiVersion,PRIVATE") -class mcp_Tool_DrawCircleCpp : public mcp_Tool +namespace +{ + // 辅助函数:Base64 编码 + std::string Base64Encode(const unsigned char *data, size_t length) + { + static const char encoding_table[] = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; + size_t out_len = 4 * ((length + 2) / 3); + std::string ret(out_len, '\0'); + size_t i; + char *p = const_cast(ret.c_str()); + + for (i = 0; i < length - 2; i += 3) + { + *p++ = encoding_table[(data[i] >> 2) & 0x3F]; + *p++ = encoding_table[((data[i] & 0x3) << 4) | ((int)(data[i + 1] & 0xF0) >> 4)]; + *p++ = encoding_table[((data[i + 1] & 0xF) << 2) | ((int)(data[i + 2] & 0xC0) >> 6)]; + *p++ = encoding_table[data[i + 2] & 0x3F]; + } + if (i < length) + { + *p++ = encoding_table[(data[i] >> 2) & 0x3F]; + if (i == (length - 1)) + { + *p++ = encoding_table[((data[i] & 0x3) << 4)]; + *p++ = '='; + } + else + { + *p++ = encoding_table[((data[i] & 0x3) << 4) | ((int)(data[i + 1] & 0xF0) >> 4)]; + *p++ = encoding_table[((data[i + 1] & 0xF) << 2)]; + } + *p++ = '='; + } + return ret; + } + + // 辅助函数:获取 GDI+ 图像编码器的 Clsid (用于转 PNG) + int GetEncoderClsid(const WCHAR *format, CLSID *pClsid) + { + UINT num = 0, size = 0; + Gdiplus::GetImageEncodersSize(&num, &size); + if (size == 0) return -1; + Gdiplus::ImageCodecInfo *pImageCodecInfo = (Gdiplus::ImageCodecInfo *)(malloc(size)); + if (pImageCodecInfo == NULL) return -1; + Gdiplus::GetImageEncoders(num, size, pImageCodecInfo); + for (UINT j = 0; j < num; ++j) + { + if (_tcscmp(pImageCodecInfo[j].MimeType, format) == 0) + { + *pClsid = pImageCodecInfo[j].Clsid; + free(pImageCodecInfo); + return j; + } + } + free(pImageCodecInfo); + return -1; + } +} + +class mcp_Tool_ViewportScreenshot: public mcp_Tool { public: virtual nlohmann::json Execute(const nlohmann::json &args) override { - // 1. 从大模型传入的参数中提取坐标和半径 - double x = args.value("cx", 0.0); - double y = args.value("cy", 0.0); - double r = args.value("r", 100.0); + // 1. 获取 AutoCAD 当前文档的绘图区句柄 + HWND hWnd = adsw_acadDocWnd(); + if (!hWnd) + return mcp_Tool_Utility::make_error(-32000, L"无法获取CAD绘图区窗口句柄"); - // 2. 获取当前活动文档 - AcApDocument* pDoc = acDocManager->curDocument(); - if (pDoc == nullptr) + // 2. 初始化 GDI+ + Gdiplus::GdiplusStartupInput gdiplusStartupInput; + ULONG_PTR gdiplusToken; + Gdiplus::GdiplusStartup(&gdiplusToken, &gdiplusStartupInput, NULL); + + std::string base64Result; + bool success = false; + + // 3. 开始 Windows GDI 抓屏逻辑 (使用花括号限定生命周期) { - return { { "error", { { "code", -32000 }, { "message", "当前没有打开的 CAD 文档" } } } }; - } + RECT rc; + GetClientRect(hWnd, &rc); + int width = rc.right - rc.left; + int height = rc.bottom - rc.top; - // 3. 极其重要:因为调用发起者是隐式窗口消息,并非标准 CAD 命令,必须显式锁文档! - acDocManager->lockDocument(pDoc); - AcDbDatabase* pDb = pDoc->database(); + HDC hdcScreen = GetDC(hWnd); + HDC hdcMem = CreateCompatibleDC(hdcScreen); + HBITMAP hBitmap = CreateCompatibleBitmap(hdcScreen, width, height); + HBITMAP hOldBitmap = (HBITMAP)SelectObject(hdcMem, hBitmap); - AcDbBlockTable* pBlockTable = nullptr; - Acad::ErrorStatus es = pDb->getBlockTable(pBlockTable, AcDb::kForRead); - - if (es == Acad::eOk) - { - AcDbBlockTableRecord* pModelSpace = nullptr; - es = pBlockTable->getAt(ACDB_MODEL_SPACE, pModelSpace, AcDb::kForWrite); - - if (es == Acad::eOk) + // 将屏幕内容块拷贝到内存位图中 + BitBlt(hdcMem, 0, 0, width, height, hdcScreen, 0, 0, SRCCOPY); + + // 将 GDI 位图转换为 GDI+ 位图 + Gdiplus::Bitmap bmp(hBitmap, NULL); + + // 创建内存流,将位图以 PNG 格式保存到内存,而非硬盘 + IStream *pStream = nullptr; + if (CreateStreamOnHGlobal(NULL, TRUE, &pStream) == S_OK) { - // 在底层 C++ 层面构造实体对象 - AcGePoint3d center(x, y, 0.0); - AcGeVector3d normal(0.0, 0.0, 1.0); // Z轴法线 - AcDbCircle* pCircle = new AcDbCircle(center, normal, r); + CLSID pngClsid; + GetEncoderClsid(L"image/png", &pngClsid); - // 追加到模型空间 - AcDbObjectId circleId; - pModelSpace->appendAcDbEntity(circleId, pCircle); - - // 释放对象 - pCircle->close(); - pModelSpace->close(); + if (bmp.Save(pStream, &pngClsid, NULL) == Gdiplus::Ok) + { + // 从内存流中读取二进制数据 + STATSTG stat; + pStream->Stat(&stat, STATFLAG_NONAME); + ULONG size = stat.cbSize.LowPart; + + std::vector buffer(size); + LARGE_INTEGER liZero = {}; + pStream->Seek(liZero, STREAM_SEEK_SET, NULL); + + ULONG bytesRead; + pStream->Read(buffer.data(), size, &bytesRead); + + // 转换为 Base64 + base64Result = Base64Encode(buffer.data(), size); + success = true; + } + pStream->Release(); } - pBlockTable->close(); + + // 清理 GDI 资源 + SelectObject(hdcMem, hOldBitmap); + DeleteObject(hBitmap); + DeleteDC(hdcMem); + ReleaseDC(hWnd, hdcScreen); } - // 4. 解锁文档并刷新显示 - acDocManager->unlockDocument(pDoc); - acedUpdateDisplay(); + // 4. 关闭 GDI+ + Gdiplus::GdiplusShutdown(gdiplusToken); - // 5. 返回标准 MCP 结果,如果失败可以根据 es 的值返回 make_error - if (es == Acad::eOk) + // 5. 按照 MCP 多模态协议返回数据 + if (success) { - return mcp_Tool_Utility::make_text_result(L"C++底层接口已成功在模型空间生成圆形!"); + return mcp_Tool_Utility::make_mixed_result(L"已成功截取 AutoCAD 当前视口画面。", base64Result, "image/png"); } else { - return mcp_Tool_Utility::make_error((int)es, L"ObjectARX 内部错误"); + return mcp_Tool_Utility::make_error(-32001, L"截图或图像编码失败"); } } }; -// 使用宏导出 C 风格工厂函数,供基座的 LoadLibrary 解析 -EXPORT_MCP_TOOL(mcp_Tool_DrawCircleCpp, CreateDrawCircleToolCpp) +EXPORT_MCP_TOOL(mcp_Tool_ViewportScreenshot, CreateViewportScreenshotTool) diff --git a/test.py b/test.py new file mode 100644 index 0000000..76bd46c --- /dev/null +++ b/test.py @@ -0,0 +1,41 @@ +import socket +import json + +def test_fake_llm_zoom(): + print("=== 模拟 LLM 发起缩放指令测试 ===") + + try: + client = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + client.connect(("127.0.0.1", 8080)) + + # 伪造大模型生成的 tool_call 报文 + req = { + "jsonrpc": "2.0", + "id": 999, + "method": "tools/call", + "params": { + "name": "zoom_window_normalized", + "arguments": { + "x1": 0.4, + "y1": 0.4, + "x2": 0.6, + "y2": 0.6 + } + } + } + + print(f"\n[发送] 请求参数:\n{json.dumps(req['params'], indent=2)}") + client.sendall(json.dumps(req).encode('utf-8')) + + response_data = client.recv(4096) + + print(f"\n[接收] CAD 基座返回:") + print(json.dumps(json.loads(response_data.decode('utf-8')), indent=2, ensure_ascii=False)) + + client.close() + + except Exception as e: + print(f"测试失败: {e}") + +if __name__ == "__main__": + test_fake_llm_zoom() \ No newline at end of file diff --git a/test_autoagent.py b/test_autoagent.py index a71c278..0241a2f 100644 --- a/test_autoagent.py +++ b/test_autoagent.py @@ -1,64 +1,43 @@ import socket import json -import time +import base64 -def send_request(req_data, silent=False): - try: - client = socket.socket(socket.AF_INET, socket.SOCK_STREAM) - client.connect(("127.0.0.1", 8080)) - - # 发送 UTF-8 编码的 JSON 请求 - client.sendall(json.dumps(req_data).encode('utf-8')) - - # 接收并解码结果 - response = client.recv(4096) - - if not silent: - print(">>> 收到回复:") - print(json.dumps(json.loads(response.decode('utf-8')), indent=2, ensure_ascii=False)) - print("-" * 50) - - client.close() - except Exception as e: - print(f"连接失败: {e}") - -if __name__ == "__main__": - print("=== 测试 1: 验证 C++ DLL 工具调用 ===") - request_call = { +def test_screenshot(): + client = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + client.connect(("127.0.0.1", 8080)) + + req = { "jsonrpc": "2.0", - "id": 4, + "id": 5, "method": "tools/call", "params": { - "name": "draw_circle_cpp", - "arguments": { - "cx": 2000.0, - "cy": 2000.0, - "r": 500.0 - } + "name": "get_viewport_screenshot", + "arguments": {} } } - send_request(request_call) + client.sendall(json.dumps(req).encode('utf-8')) - # 稍作停顿,方便观察 CAD 屏幕 - time.sleep(1) + # 图片 base64 报文可能很大,需要循环接收直到拿完 + response_data = b"" + while True: + chunk = client.recv(8192) + response_data += chunk + if len(chunk) < 8192: + break + + client.close() - print("\n=== 测试 2: C++ 并发写入压力测试 (生成10个同心圆) ===") - for i in range(10): - radius = 100.0 + i * 50.0 - req = { - "jsonrpc": "2.0", - "id": 100 + i, - "method": "tools/call", - "params": { - "name": "draw_circle_cpp", - "arguments": { - "cx": 4000.0, - "cy": 2000.0, - "r": radius - } - } - } - # 连续快速发送,不打印详细返回以模拟极限并发 - send_request(req, silent=True) + resp_json = json.loads(response_data.decode('utf-8')) - print(">>> 并发指令发送完毕,请检查 CAD 屏幕。") \ No newline at end of file + # 解析并保存图片 + content_list = resp_json["result"]["content"] + for item in content_list: + if item["type"] == "image": + base64_data = item["data"] + image_bytes = base64.b64decode(base64_data) + with open("test_screenshot.png", "wb") as f: + f.write(image_bytes) + print("截图已保存为 test_screenshot.png,请在当前目录查看!") + +if __name__ == "__main__": + test_screenshot() \ No newline at end of file diff --git a/test_screenshot.png b/test_screenshot.png new file mode 100644 index 0000000..fefc0ba Binary files /dev/null and b/test_screenshot.png differ