<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/amie-video-以三代理拆分即時問診-模擬臨床評分達-83-fbacf710</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T16:02:24.599Z</news:publication_date>
      <news:title>AMIE Video 以三代理拆分即時問診，模擬臨床評分達 83%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/布局匹配取代直接座標生成-gui-定位在-screenspot-pro-達-41-3-b6f27185</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T16:03:18.324Z</news:publication_date>
      <news:title>布局匹配取代直接座標生成，GUI 定位在 ScreenSpot-Pro 達 41.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/open-ea-將視覺生成評測代理縮至-3b-但目前只完整支援-vbench-影片流程-344ed0ae</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T14:04:11.444Z</news:publication_date>
      <news:title>Open-EA 將視覺生成評測代理縮至 3B，但目前只完整支援 VBench 影片流程</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/tide-修正教師-學生錯配-qwen3-推理-avg-8-由-6-9-升至-20-3-d58973e0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T14:03:05.897Z</news:publication_date>
      <news:title>TIDE 修正教師—學生錯配，Qwen3 推理 Avg@8 由 6.9% 升至 20.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/icbq-重查量化區塊接縫-qwen3-8b-的三值模型困惑度由-5752-9-降至-29-9-445d60f4</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T12:02:53.968Z</news:publication_date>
      <news:title>ICBQ 重查量化區塊接縫，Qwen3‑8B 的三值模型困惑度由 5752.9 降至 29.9</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/elasticback-在單一-agent-skill-埋入條件式後門-多數模型誤觸率壓至-2-以下-2c32b87f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T12:02:06.063Z</news:publication_date>
      <news:title>ElasticBack 在單一 Agent Skill 埋入條件式後門，多數模型誤觸率壓至 2% 以下</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/queryproof-以規則閘門攔截錯誤-sql-7b-代理的-business-truth-rate-達-56-2-79ec5be5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T10:03:57.758Z</news:publication_date>
      <news:title>QueryProof 以規則閘門攔截錯誤 SQL，7B 代理的 Business Truth Rate 達 56.2%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/matryoshka-將三種模型嵌入單一權重-訓練算力減少-36-8401469b</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T10:02:49.281Z</news:publication_date>
      <news:title>Matryoshka 將三種模型嵌入單一權重，訓練算力減少 36%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/p3-先共同規劃程式與證明-lean-驗證解題率提高-4-6-至-11-2-個百分點-8a1fadc3</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T08:03:31.799Z</news:publication_date>
      <news:title>P³ 先共同規劃程式與證明，Lean 驗證解題率提高 4.6 至 11.2 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/archagent-v2-以代理演化三層預取器-dpc4-模擬效能略勝人工冠軍-41a53477</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T08:02:39.577Z</news:publication_date>
      <news:title>ArchAgent v2 以代理演化三層預取器，DPC4 模擬效能略勝人工冠軍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/tevatron-elastic-用單一檢索-checkpoint-調節深度-token-與向量寬度-15bcee57</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T06:05:18.646Z</news:publication_date>
      <news:title>Tevatron-Elastic 用單一檢索 checkpoint 調節深度、token 與向量寬度</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/swe-bench-promax-將重構測試擴至七種語言-最佳代理仍只解出-41-2-b11e109e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T06:03:54.282Z</news:publication_date>
      <news:title>SWE-Bench ProMax 將重構測試擴至七種語言，最佳代理仍只解出 41.2%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/kvdiagnosis-逐筆追查-kv-快取壓縮失敗-公開-12-520-筆正確轉錯誤案例-e8ff16bf</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T04:03:22.728Z</news:publication_date>
      <news:title>KVDiagnosis 逐筆追查 KV 快取壓縮失敗，公開 12,520 筆正確轉錯誤案例</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/she-從代理執行軌跡改寫安全外殼-攻擊成功率由-17-1-降至-5-5-0156d4a3</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T04:02:32.914Z</news:publication_date>
      <news:title>SHE 從代理執行軌跡改寫安全外殼，攻擊成功率由 17.1% 降至 5.5%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/anthropic-為新-claude-輸出加入模型級文字浮水印-檔案採-c2pa-簽章-ded4a946</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T02:04:02.163Z</news:publication_date>
      <news:title>Anthropic 為新 Claude 輸出加入模型級文字浮水印，檔案採 C2PA 簽章</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/擴散式-llm-的安全神經元可跨架構轉移-剪枝使拒答繞過率升至-86-6-2b3a8212</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T02:05:00.589Z</news:publication_date>
      <news:title>擴散式 LLM 的安全神經元可跨架構轉移，剪枝使拒答繞過率升至 86.6%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/openai-關閉-gpt-5-2-chat-latest-遷移不能只替換模型名稱-53c9242e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T00:03:20.442Z</news:publication_date>
      <news:title>OpenAI 關閉 `gpt-5.2-chat-latest`，遷移不能只替換模型名稱</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/laguna-s-2-1-int4-搭-dflash-四張-rtx-3090-實測-200k-上下文與峰值-282-5-token-s-4929d332</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-11T00:04:18.830Z</news:publication_date>
      <news:title>Laguna S 2.1 INT4 搭 DFlash，四張 RTX 3090 實測 200K 上下文與峰值 282.5 token/s</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ptq4snn-同時量化權重與膜電位-事件分類僅損失-1-個百分點-fe63772d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T22:04:26.922Z</news:publication_date>
      <news:title>PTQ4SNN 同時量化權重與膜電位，事件分類僅損失 1 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/skillprox-驗證再收斂代理技能-qwen3-6-27b-表格任務準確率升至-54-5-9516b866</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T22:03:08.928Z</news:publication_date>
      <news:title>SkillProx 驗證再收斂代理技能，Qwen3.6-27B 表格任務準確率升至 54.5%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/docmemo-以貝氏頁面記憶重查長文件-mmlongbench-準確率達-71-3-b0fde8e2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T20:03:44.115Z</news:publication_date>
      <news:title>DocMemo 以貝氏頁面記憶重查長文件，MMLongBench 準確率達 71.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/coba-動態分配推理算力-以少-58-9-加權-token-追平-best-of-16-ef92bc45</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T20:02:17.700Z</news:publication_date>
      <news:title>CoBa 動態分配推理算力，以少 58.9% 加權 token 追平 Best-of-16</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/openai-限定釋出-gpt-5-6-cyber-高風險請求完成率由-1-5-升至-95-983b7cb3</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T18:03:52.227Z</news:publication_date>
      <news:title>OpenAI 限定釋出 GPT‑5.6‑Cyber：高風險請求完成率由 1.5% 升至 95%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/meta-開放-muse-glimmer-30b-17gb-量化版搭-dflash-在-rtx-5090-達-233-token-s-19074787</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T18:02:51.765Z</news:publication_date>
      <news:title>Meta 開放 Muse Glimmer 30B：17GB 量化版搭 DFlash 在 RTX 5090 達 233 token/s</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/a2e-統一-23-項基準與-9-種代理外殼-單格評測仍只有-5-題-1280b64a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T16:03:00.116Z</news:publication_date>
      <news:title>A²E 統一 23 項基準與 9 種代理外殼，單格評測仍只有 5 題</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/tepa-撤銷過期代理記憶-反轉任務成功率由-21-0-升至-95-0-20f662e4</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T16:03:54.605Z</news:publication_date>
      <news:title>TEPA 撤銷過期代理記憶，反轉任務成功率由 21.0% 升至 95.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/finrank-用-6-021-個混淆段落測金融-rag-7b-嵌入模型-recall-10-僅-44-8-1242ff21</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T14:04:16.101Z</news:publication_date>
      <news:title>FinRank 用 6,021 個混淆段落測金融 RAG，7B 嵌入模型 Recall@10 僅 44.8%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/orcetra-自揭-automl-基準失真-同一子集勝率由-59-4-降至-34-3-8723a1a0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T14:02:52.995Z</news:publication_date>
      <news:title>Orcetra 自揭 AutoML 基準失真：同一子集勝率由 59.4% 降至 34.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/residencyrl-以最長-60-輪模擬問診訓練臨床代理-對抗案例診斷率升至-88-90f0c6be</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T12:03:25.136Z</news:publication_date>
      <news:title>ResidencyRL 以最長 60 輪模擬問診訓練臨床代理，對抗案例診斷率升至 88%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/muon-加速-grokking-後仍會失效-九組設定全數出現泛化崩落-10d7d82c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T12:02:26.557Z</news:publication_date>
      <news:title>Muon 加速 grokking 後仍會失效：九組設定全數出現泛化崩落</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/shepherd-v0-3-0-將代理修改留作可審核提案-但論文展示的自動分叉監督仍未完整交付-b3de0019</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T10:04:52.541Z</news:publication_date>
      <news:title>Shepherd v0.3.0 將代理修改留作可審核提案，但論文展示的自動分叉監督仍未完整交付</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/muse-code-預設載入-claude-與-codex-個人指令-跨供應商相容性引出資料邊界問題-6b58bbe6</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T10:03:38.186Z</news:publication_date>
      <news:title>Muse Code 預設載入 Claude 與 Codex 個人指令，跨供應商相容性引出資料邊界問題</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cellworld-改以潛在細胞表徵預訓練-5-74m-模型在-18-項空間轉錄體任務勝過所列基線-a17dc5c4</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T06:03:48.608Z</news:publication_date>
      <news:title>CellWorld 改以潛在細胞表徵預訓練，5.74M 模型在 18 項空間轉錄體任務勝過所列基線</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/agentpatch-免訓練修補多模態代理合併-六項基準平均分由-54-5-升至-56-6-0f6fce76</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T06:02:36.585Z</news:publication_date>
      <news:title>AgentPatch 免訓練修補多模態代理合併，六項基準平均分由 54.5 升至 56.6</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/open-science-v0-12-1-將科研代理的壓縮-執行與產物證據留在本機工作區-73248d22</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T04:03:31.139Z</news:publication_date>
      <news:title>Open Science v0.12.1 將科研代理的壓縮、執行與產物證據留在本機工作區</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/openai-暫停部分-astra-開發-初測無法排除模型已達-critical-網攻能力-26d5d557</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T04:02:43.008Z</news:publication_date>
      <news:title>OpenAI 暫停部分 Astra 開發：初測無法排除模型已達「Critical」網攻能力</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/didpo-把程式-diff-拆成信用單元-qwen2-5-coder-7b-主要評測平均升至-48-4-dd2b8db5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T02:04:45.548Z</news:publication_date>
      <news:title>DiDPO 把程式 diff 拆成信用單元，Qwen2.5-Coder-7B 主要評測平均升至 48.4%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/fisher-r1-以可驗證-p-值訓練統計代理-p-hard-嚴格-pass-1-達-33-0-8926a0a0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T02:03:21.063Z</news:publication_date>
      <news:title>Fisher-R1 以可驗證 p 值訓練統計代理，P-Hard 嚴格 pass@1 達 33.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ling-3-0-flash-int4-單機實測達-38-7-token-s-錯用主線-vllm-可能無聲產生錯誤輸出-cda07d6e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T00:04:10.660Z</news:publication_date>
      <news:title>Ling-3.0-flash INT4 單機實測達 38.7 token/s，錯用主線 vLLM 可能無聲產生錯誤輸出</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/minimax-開放-h3-基礎權重-33b-單流模型同步生成最長-15-秒立體聲影片-ee992bda</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T00:03:16.651Z</news:publication_date>
      <news:title>MiniMax 開放 H3 基礎權重：33B 單流模型同步生成最長 15 秒立體聲影片</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/harnessopt-bench-測試模型改寫代理外殼-模型選擇的影響約為編碼工具-1-8-倍-76ad33a2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T18:01:56.310Z</news:publication_date>
      <news:title>HarnessOpt-Bench 測試模型改寫代理外殼，模型選擇的影響約為編碼工具 1.8 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/mist-測出-23-款模型皆受錯誤提示影響-scope-將-qwen3-4b-答案翻轉率由-35-0-降至-16-3-7d5c379a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T22:03:54.150Z</news:publication_date>
      <news:title>MIST 測出 23 款模型皆受錯誤提示影響，SCOPE 將 Qwen3-4B 答案翻轉率由 35.0% 降至 16.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ts-rag-以參考-token-融合歷史序列-六項預測資料集平均-mse-降至-0-310-f32d2c7e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T20:02:35.854Z</news:publication_date>
      <news:title>TS-RAG 以參考 token 融合歷史序列，六項預測資料集平均 MSE 降至 0.310</news:title>
    </news:news>
  </url>
</urlset>
