<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>AI Tool Stack — Clinical LLM Benchmarks &amp; AI Tools for Surgeons</title>
    <link>https://tanhaosheng.asia/</link>
    <description>Hands-on, source-backed AI tool comparisons and clinical LLM benchmarks by Dr. Tan Haosheng (谭好升), Associate Chief Physician, Taizhou People's Hospital.</description>
    <language>en</language>
    <lastBuildDate>Fri, 31 Jul 2026 12:00:00 +0000</lastBuildDate>
    <atom:link href="https://tanhaosheng.asia/feed.xml" rel="self" type="application/rss+xml"/>
    <item>
      <title>About - Dr. Tan Haosheng · AI Tool Stack</title>
      <link>https://tanhaosheng.asia/about</link>
      <guid>https://tanhaosheng.asia/about</guid>
      <description>About Dr. Tan Haosheng (谭好升): Associate Chief Physician, MD/PhD from Tsinghua–Peking Union / National Cancer Center, now at Taizhou People's Hospital. Author of AI Tool Stack.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Same Case, Web vs Agent: 4 Models, 12 Runs (Review 3/5) | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agent-vs-web-isolation</link>
      <guid>https://tanhaosheng.asia/agent-vs-web-isolation</guid>
      <description>Part 3: 4 models × 3 runs on one breast-cancer case — web vs WorkBuddy agent. Agent raises the floor, not the ceiling (16-item/100 rubric).</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Configuring Agnes in WorkBuddy: A Free Agent Model — Hands-on | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agnes-workbuddy-config</link>
      <guid>https://tanhaosheng.asia/agnes-workbuddy-config</guid>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>ChatGLM 5.2 vs DeepSeek on a Breast Cancer Case | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/chatglm-vs-deepseek-clinical-hands-on</link>
      <guid>https://tanhaosheng.asia/chatglm-vs-deepseek-clinical-hands-on</guid>
      <description>A surgeon's hands-on test of ChatGLM 5.2 vs DeepSeek on a real anonymized post-op breast-cancer case, scored with the 16-item/100 rubric against 2026 CBCS. ChatGLM 5/100; DeepSeek 28/100.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>4 Free AI Models on One Breast Cancer Case | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/four-models-clinical-showdown</link>
      <guid>https://tanhaosheng.asia/four-models-clinical-showdown</guid>
      <description>4 free AI models on one anonymized post-op breast-cancer case, scored with the 16-item/100 rubric. Only DeepSeek V4 Pro caught the staging trap (50/100); Doubao 41; DeepSeek web 28; ChatGLM 5.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>I Published 361 AI Articles, Then Deleted 358 — How This Site Was Built</title>
      <link>https://tanhaosheng.asia/how-i-built-this-site-with-ai</link>
      <guid>https://tanhaosheng.asia/how-i-built-this-site-with-ai</guid>
      <description>A surgeon's honest build log: 4 days from idea to launch with an AI agent — the content-farm trap, the great deletion (361 → 3 articles), the pivot to real hands-on testing, and every lesson learned.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>AI Tool Stack — Clinical LLM Benchmark &amp; AI Tools for Surgeons</title>
      <link>https://tanhaosheng.asia/</link>
      <guid>https://tanhaosheng.asia/</guid>
      <description>The AI-era tool-selection guide for surgeons — pick the right AI model and tool with clear, source-backed comparisons.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Kimi K2.6: Web vs WorkBuddy Agent on One Case (Review 4/5) | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/kimi-k2-6-web-vs-agent</link>
      <guid>https://tanhaosheng.asia/kimi-k2-6-web-vs-agent</guid>
      <description>Kimi K2.6 tested as plain web chat vs inside the WorkBuddy agent on one anonymized post-op breast-cancer case. Agent median 72.2/100 vs web 68.9/100 (16-item rubric).</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Model Reviews: Free AI Hands-on Tests (V1→V5) | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/model-reviews</link>
      <guid>https://tanhaosheng.asia/model-reviews</guid>
      <description>Free AI models tested hands-on on one anonymized post-op breast-cancer case, scored with a 16-item/100 rubric against 2026 CBCS guidelines. The full V1→V5 series.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Site Map — AI Tool Stack</title>
      <link>https://tanhaosheng.asia/sitemap</link>
      <guid>https://tanhaosheng.asia/sitemap</guid>
      <description>Full index of all pages on AI Tool Stack — free AI model hands-on tests and clinical LLM benchmarks by Dr. Tan Haosheng.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>全系列统一评分标准 · V1–V5 重算（单一病例）</title>
      <link>https://tanhaosheng.asia/unified-scoring-v1-v5</link>
      <guid>https://tanhaosheng.asia/unified-scoring-v1-v5</guid>
      <description>Unified scoring for the V1–V5 clinical LLM series: one patient, one 16-item/100 rubric, two prompt modes. Retired the old 10-item table and the cross-table '+39' delta. Full leaderboards and raw-data links.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>指南注入实测 · V5（四模型同尺对比）</title>
      <link>https://tanhaosheng.asia/v5-guideline-report</link>
      <guid>https://tanhaosheng.asia/v5-guideline-report</guid>
      <description>V5 hands-on: guideline injection tested on four free models against one anonymized post-op breast-cancer case with the unified 16-item/100 rubric. Baseline 60–68, +6 to +10 after guidance.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Running a Free Agent on WorkBuddy (Hunyuan hy3): Hands-on Log | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/workbuddy-free-agent-hunyuan-hy3-en</link>
      <guid>https://tanhaosheng.asia/workbuddy-free-agent-hunyuan-hy3-en</guid>
      <description>A screenshot-driven walkthrough by Tan Haosheng on WorkBuddy (Tencent): sign-up, WeChat login, the credits system, where Hunyuan hy3 fits, and the model-picker UI.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>WorkBuddy 免费运行 Agent（混元 hy3）实测：注册、积分机制与首批流程 - AI Tool Stack</title>
      <link>https://tanhaosheng.asia/workbuddy-free-agent-hunyuan-hy3</link>
      <guid>https://tanhaosheng.asia/workbuddy-free-agent-hunyuan-hy3</guid>
      <description>作者 Tan Haosheng 在 WorkBuddy（workbuddy.cn）实操的全记录。详细拆解注册、微信登录、积分机制（任务/签到）、混元 hy3 模型定位与首页/下载页界面。后续段落（建 Agent / 跑任务 / 积分消耗 / 产出物）随流程补全。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Free-Model Clinical LLM Leaderboard | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/leaderboard/</link>
      <guid>https://tanhaosheng.asia/leaderboard/</guid>
      <description>A living leaderboard of free AI models on one breast-cancer case, scored against 2026 CBCS / CSCO 2024 / NCCN 2025. Key finding: the agent wrapper raises a model's floor more than its ceiling.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Case 1 — Reading the data, checking a draft · AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agents-for-research/case-1</link>
      <guid>https://tanhaosheng.asia/agents-for-research/case-1</guid>
      <description>A surgeon's real hands-on test of an AI agent on a research workload: identify a sample preparation, then hold a draft to the data. Behaviour and timing only — no scientific result is disclosed.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Case 2 — The mismatch trap · AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agents-for-research/case-2</link>
      <guid>https://tanhaosheng.asia/agents-for-research/case-2</guid>
      <description>A surgeon's real hands-on test of an AI agent: attach a draft from an unrelated line of work to the same dataset and see whether the agent notices the mismatch. Behaviour and timing only.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Case 3 — From raw data to a research plan · AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agents-for-research/case-3</link>
      <guid>https://tanhaosheng.asia/agents-for-research/case-3</guid>
      <description>A surgeon's real hands-on test of an AI agent as a forward planner: given raw data, what could realistically be written and where could it go? Behaviour and timing only — no result disclosed.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>AI Agents for Medical Research — Surgeon-Tested | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agents-for-research/</link>
      <guid>https://tanhaosheng.asia/agents-for-research/</guid>
      <description>First-hand case studies of an AI agent (WorkBuddy Hy3) on a surgeon-researcher's own workload — what it handled, where it pushed back, where it failed. Behaviour only; no results disclosed.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Methodology: Testing AI Agents on Medical Research | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/agents-for-research/methodology</link>
      <guid>https://tanhaosheng.asia/agents-for-research/methodology</guid>
      <description>The protocol for testing AI agents on real medical research tasks: prompt construction, data attachment, evaluation rubric, and the four failure modes to watch. Reproducible and honest.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>关于 — 谭好升医生 · AI 工具栈</title>
      <link>https://tanhaosheng.asia/zh/about</link>
      <guid>https://tanhaosheng.asia/zh/about</guid>
      <description>关于谭好升医生（副主任医师、医学博士，清华-协和/国家癌症中心，现于泰州市人民医院）。AI 工具栈作者：面向临床与科研、有真实一手实测的 AI 工具选型指南。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>利益披露 — AI Tool Stack</title>
      <link>https://tanhaosheng.asia/zh/affiliate-disclosure</link>
      <guid>https://tanhaosheng.asia/zh/affiliate-disclosure</guid>
      <description>AI Tool Stack 如何处理联盟（affiliate）与推广链接。部分页面可能含推广链接，但不会影响你的价格，也不会改变我们独立的评测结论。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>在 WorkBuddy 中配置 Agnes：免费智能体模型（稍慢但可用）—— 实操指南</title>
      <link>https://tanhaosheng.asia/zh/agnes-workbuddy-config</link>
      <guid>https://tanhaosheng.asia/zh/agnes-workbuddy-config</guid>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>4 个免费 AI 模型同测一个乳腺癌病例 | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/zh/four-models-clinical-showdown</link>
      <guid>https://tanhaosheng.asia/zh/four-models-clinical-showdown</guid>
      <description>一位外科医生的扩展实测：4 个免费 AI 模型跑同一个脱敏术后乳腺癌病例，用同一套 16 项/100 分标准对照 2026 CBCS 指南评分。仅 DeepSeek V4 Pro 识破分期陷阱（50/100）。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>我 2 天生成了 361 篇 AI 文章，然后删掉了 358 篇 —— 这个网站真实的建站过程</title>
      <link>https://tanhaosheng.asia/zh/how-i-built-this-site-with-ai</link>
      <guid>https://tanhaosheng.asia/zh/how-i-built-this-site-with-ai</guid>
      <description>一名外科医生的真实建站日志：4 天时间靠 AI 智能体从构思到上线——内容农场的坑、361 篇删到 3 篇的大删减、转向真实实测的全过程与全部教训。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>AI Tool Stack — 🛠️ AI 工具栈</title>
      <link>https://tanhaosheng.asia/zh/</link>
      <guid>https://tanhaosheng.asia/zh/</guid>
      <description>面向外科医生的 AI 时代选型指南——挑选合适的 AI 模型与工具，清晰、有来源对比。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>智谱清言 GLM 5.2 网页版 · 三段完整回答（外部附录）</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/chatglm_answers</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/chatglm_answers</guid>
      <description>ChatGLM 5.2 web three full runs (raw text) from the clinical LLM benchmark, for item-by-item scoring verification.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>ChatGLM 5.2 · 临床决策实测分析报告（含 DeepSeek 对比）</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/chatglm_report</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/chatglm_report</guid>
      <description>ChatGLM 5.2 · 临床决策实测分析报告（含 DeepSeek 对比）。谭好升（副主任医师，泰州市人民医院）主导的临床大模型实测。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>ChatGLM 5.2 Clinical Benchmark Report (vs DeepSeek) | AI Tool Stack</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/chatglm_report_en</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/chatglm_report_en</guid>
      <description>ChatGLM 5.2 · Clinical Benchmark Report (with DeepSeek comparison). Clinical LLM benchmarking for surgeons by Tan Haosheng, MD PhD.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>DeepSeek 快速模式 · 三段完整回答（外部附录）</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/deepseek_answers</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/deepseek_answers</guid>
      <description>DeepSeek web Fast mode three full runs (raw text) from the clinical LLM benchmark, for item-by-item scoring verification.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>DeepSeek 快速模式 · 临床决策实测分析报告</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/deepseek_report</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/deepseek_report</guid>
      <description>DeepSeek 快速模式 · 临床决策实测分析报告。谭好升（副主任医师，泰州市人民医院）主导的临床大模型实测。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>DeepSeek Fast Mode · Clinical Benchmark Report</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/deepseek_report_en</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/deepseek_report_en</guid>
      <description>DeepSeek Fast Mode · Clinical Benchmark Report. Clinical LLM benchmarking for surgeons by Tan Haosheng, MD PhD.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>LLM 临床决策实测框架 v1.0</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/framework</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/framework</guid>
      <description>中文 LLM 临床决策实测框架 v1.0：标准化 Prompt、16 项评分 Rubric、参考答案要点、复测稳定性要求。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Clinical LLM Benchmark Framework v1.0 (EN)</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/framework_en</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/framework_en</guid>
      <description>English clinical LLM benchmark framework v1.0: standardized prompt, 16-item scoring rubric, answer key, re-test stability rules.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>中文 LLM 临床决策实测 · 总入口</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/</guid>
      <description>中文 LLM 临床决策实测 · 总入口。谭好升（副主任医师，泰州市人民医院）主导的临床大模型实测。</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
    <item>
      <title>Clinical LLM Benchmark · Hub</title>
      <link>https://tanhaosheng.asia/clinical-benchmark/index_en</link>
      <guid>https://tanhaosheng.asia/clinical-benchmark/index_en</guid>
      <description>Clinical LLM Benchmark · Hub. Clinical LLM benchmarking for surgeons by Tan Haosheng, MD PhD.</description>
      <pubDate>Fri, 31 Jul 2026 12:00:00 +0000</pubDate>
    </item>
  </channel>
</rss>