<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/mist-以四種配對情境測試-選擇性信任-誤導訊息令-23-款模型平均掉-17-1-分-06e7fd0d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T14:02:49.637Z</news:publication_date>
      <news:title>MIST 以四種配對情境測試「選擇性信任」，誤導訊息令 23 款模型平均掉 17.1 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/程式化工具呼叫在-14-款模型中有-11-款不輸-json-長鏈任務差距達-18-8-個百分點-9fc48de7</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T14:03:52.237Z</news:publication_date>
      <news:title>程式化工具呼叫在 14 款模型中有 11 款不輸 JSON，長鏈任務差距達 18.8 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/geniworld-將機器人動作渲染成視覺條件-合成資料使實機成功率升至-69-0-1afe56be</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T12:03:40.656Z</news:publication_date>
      <news:title>GeniWorld 將機器人動作渲染成視覺條件，合成資料使實機成功率升至 69.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/calibforge-以求解器分歧校準終端任務-跨基準最高提升-30-04-個百分點-e98cc315</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T12:02:37.276Z</news:publication_date>
      <news:title>CalibForge 以求解器分歧校準終端任務，跨基準最高提升 30.04 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/2-190-段受控影片揭露-vlm-計數邊界-增加取樣幀仍無法保證事件軌跡正確-8c9aa159</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T10:03:44.060Z</news:publication_date>
      <news:title>2,190 段受控影片揭露 VLM 計數邊界：增加取樣幀仍無法保證事件軌跡正確</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/harnessopt-bench-將代理外殼最佳化變成評測-模型選擇影響約為-coding-harness-的-1-8-倍-5badd8e6</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T10:02:29.324Z</news:publication_date>
      <news:title>HarnessOpt-Bench 將代理外殼最佳化變成評測，模型選擇影響約為 coding harness 的 1.8 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/trajdebug-追蹤代理錯誤生命週期-失敗診斷讓重跑成功率平均提高-10-8-5618b18c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T08:02:36.276Z</news:publication_date>
      <news:title>TrajDebug 追蹤代理錯誤生命週期，失敗診斷讓重跑成功率平均提高 10.8%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/microevo-以-llm-與-mcts-搜尋處理器設計-pareto-品質最高提高-36-2-63c949bb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T08:03:42.233Z</news:publication_date>
      <news:title>MicroEvo 以 LLM 與 MCTS 搜尋處理器設計，Pareto 品質最高提高 36.2%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/padoc-以版面分支並行解析文件-單張-a800-吞吐量最高提高-118-f3a71df2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T06:03:50.115Z</news:publication_date>
      <news:title>PaDoc 以版面分支並行解析文件，單張 A800 吞吐量最高提高 118%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/視覺模型呼叫裁切工具不等於使用證據-六模型增益集中於少數有效軌跡-14b25b93</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T06:02:16.012Z</news:publication_date>
      <news:title>視覺模型呼叫裁切工具不等於使用證據，六模型增益集中於少數有效軌跡</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/envace-讓單一模型同時扮演代理與環境-三項工具評測綜合分升至-32-91-f6be87e8</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T04:02:13.763Z</news:publication_date>
      <news:title>EnvACE 讓單一模型同時扮演代理與環境，三項工具評測綜合分升至 32.91%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/新研究指出-adam-會打破矩陣分解對稱性-相同函數可收斂至不同注意力表示-768bd8b8</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T00:02:25.412Z</news:publication_date>
      <news:title>新研究指出 Adam 會打破矩陣分解對稱性，相同函數可收斂至不同注意力表示</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dash-依推理軌跡動態分配蒸餾權重-qwen3-三種規模皆優於固定權重基線-b1674689</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T02:02:22.195Z</news:publication_date>
      <news:title>DASH 依推理軌跡動態分配蒸餾權重，Qwen3 三種規模皆優於固定權重基線</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/reasoning-core-以-50-種程序生成器擴充監督資料-3b-模型-drop-f1-由-33-1-升至-41-7-edf95972</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T00:02:18.768Z</news:publication_date>
      <news:title>Reasoning Core 以 50 種程序生成器擴充監督資料，3B 模型 DROP F1 由 33.1 升至 41.7</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/36-710-個-github-專案只留下-85-份代理計畫-多數尚非長期維護文件-0bfbc5ed</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T22:05:07.244Z</news:publication_date>
      <news:title>36,710 個 GitHub 專案只留下 85 份代理計畫，多數尚非長期維護文件</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/meta-推出-muse-code-以持久背景代理與事件日誌支撐長時程程式任務-0dce067c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T22:03:54.872Z</news:publication_date>
      <news:title>Meta 推出 Muse Code：以持久背景代理與事件日誌支撐長時程程式任務</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/canary-tools-將代理選錯工具拆成六類-八款模型受騙率相差約-36-倍-f68fe4af</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T20:02:57.709Z</news:publication_date>
      <news:title>Canary Tools 將代理選錯工具拆成六類，八款模型受騙率相差約 36 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/argus-讓固定權重代理累積經驗-swe-bench-pro-準確率由-59-升至約-78-7d22f77a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T20:01:58.602Z</news:publication_date>
      <news:title>Argus 讓固定權重代理累積經驗，SWE-Bench Pro 準確率由 59% 升至約 78%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/reco-以步驟獎勵協調-kv-cache-與推理長度-端到端延遲縮短逾兩倍-27b4cfe3</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T18:04:02.539Z</news:publication_date>
      <news:title>ReCo 以步驟獎勵協調 KV cache 與推理長度，端到端延遲縮短逾兩倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/contextweave-以千項可執行辦公任務重測代理記憶-偏好分數由-41-50-升至-70-60-2b64aa4b</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T18:02:38.132Z</news:publication_date>
      <news:title>ContextWeave 以千項可執行辦公任務重測代理記憶，偏好分數由 41.50 升至 70.60</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/matraix-以-1-290-維屬性合成-83-億-persona-開源百萬筆核心集測試-ai-產品-6b911261</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T16:04:00.570Z</news:publication_date>
      <news:title>MatrAIx 以 1,290 維屬性合成 83 億 persona，開源百萬筆核心集測試 AI 產品</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/azure-實測代理工作流暴露-cpu-gpu-碎片化-agora-回收閒置算力仍保住尾延遲-f3279cd7</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T16:02:51.556Z</news:publication_date>
      <news:title>Azure 實測代理工作流暴露 CPU–GPU 碎片化，Agora 回收閒置算力仍保住尾延遲</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/octolong-以跨程式庫依賴鏈訓練長上下文模型-8b-版-repoqa-得分升至-64-63-6baea572</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T14:03:05.287Z</news:publication_date>
      <news:title>OctoLong 以跨程式庫依賴鏈訓練長上下文模型，8B 版 RepoQA 得分升至 64.63</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/abseeker-從正確答案反推搜尋線索-4b-模型-browsecomp-成績升至-55-3-102b80fc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T14:03:52.503Z</news:publication_date>
      <news:title>ABSeeker 從正確答案反推搜尋線索，4B 模型 BrowseComp 成績升至 55.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/kv-cache-不等於-潛在思維-錯配快取揭露多代理增益可能只是介面效應-8c09c8d2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T12:02:25.561Z</news:publication_date>
      <news:title>KV cache 不等於「潛在思維」：錯配快取揭露多代理增益可能只是介面效應</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/小型模型自報信心難以直接控風險-22-組設定僅三組通過-20-錯誤上限-ca2d0035</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T12:03:14.692Z</news:publication_date>
      <news:title>小型模型自報信心難以直接控風險，22 組設定僅三組通過 20% 錯誤上限</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/worldcycle-用可逆動作自製驗證訊號-影片世界模型複合動作準確率提升四倍-7152550c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T10:03:22.257Z</news:publication_date>
      <news:title>WorldCycle 用可逆動作自製驗證訊號，影片世界模型複合動作準確率提升四倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/新基準發現隱性偏誤可繞過思維鏈監控-部分設定偵測率僅-5-94c057cc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T10:02:17.436Z</news:publication_date>
      <news:title>新基準發現隱性偏誤可繞過思維鏈監控，部分設定偵測率僅 5%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hive-發現語音轉錄比鍵盤錯字更傷-llm-推理-增加思考預算也未能補回-2ee0bc64</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T02:03:23.195Z</news:publication_date>
      <news:title>HIVE 發現語音轉錄比鍵盤錯字更傷 LLM 推理，增加思考預算也未能補回</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/gdpevo-以規則重組測試代理自我演化-完整提示仍領先自主學習逾-30-點-b975af80</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T02:02:14.610Z</news:publication_date>
      <news:title>GDPevo 以規則重組測試代理自我演化，完整提示仍領先自主學習逾 30 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/unlink-vl-揭露多模態遺忘落差-文字刪除難以阻止模型從圖片找回知識-622d7ea2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T00:02:59.288Z</news:publication_date>
      <news:title>UNLINK-VL 揭露多模態遺忘落差：文字刪除難以阻止模型從圖片找回知識</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/phyai-統一機器人模型雲端與邊緣推論-四類-vla-最快加速-4-65-倍-dc6d013d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T00:02:07.328Z</news:publication_date>
      <news:title>PhyAI 統一機器人模型雲端與邊緣推論，四類 VLA 最快加速 4.65 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/離線-top-k-蒸餾移除常駐教師模型-單張-h200-吞吐量最高提高-41-192aabd8</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T22:03:18.776Z</news:publication_date>
      <news:title>離線 Top-K 蒸餾移除常駐教師模型，單張 H200 吞吐量最高提高 41%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/nvidia-以線性映射跨模型轉移-kv-cache-模型切換-prefill-最快縮短至二十五分之一-f4c66071</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T22:02:08.699Z</news:publication_date>
      <news:title>NVIDIA 以線性映射跨模型轉移 KV cache，模型切換 prefill 最快縮短至二十五分之一</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ai-科學家改用賽車與牌組做前瞻評測-gpt-5-2-的-166-個構想僅命中-10-項實際創新-cef0a71e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T20:02:37.428Z</news:publication_date>
      <news:title>AI 科學家改用賽車與牌組做前瞻評測，GPT-5.2 的 166 個構想僅命中 10 項實際創新</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/factwash-在代理寫入記憶前攔截語氣漂移-否定詞跨域檢測-f1-達-0-91-c24ae81e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T20:01:35.678Z</news:publication_date>
      <news:title>FACTWASH 在代理寫入記憶前攔截語氣漂移，否定詞跨域檢測 F1 達 0.91</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/turnsight-用工具執行結果回看每一回合-qwen3-8b-三項評測平均升至-42-02-b85c8f6a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T18:02:33.620Z</news:publication_date>
      <news:title>TurnSight 用工具執行結果回看每一回合，Qwen3-8B 三項評測平均升至 42.02</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/parvl-以共享骨幹並行擴展多模態模型-視覺與語言算力可分別配置-34c53cb1</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T18:01:48.173Z</news:publication_date>
      <news:title>ParVL 以共享骨幹並行擴展多模態模型，視覺與語言算力可分別配置</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hugging-face-還原-ai-代理入侵-兩天半產生約-1-76-萬次攻擊動作-36efd6be</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T16:04:09.058Z</news:publication_date>
      <news:title>Hugging Face 還原 AI 代理入侵：兩天半產生約 1.76 萬次攻擊動作</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/agents4d-代理完成任務不代表安全-逾六成執行同時觸發危險訊號-93e0fa5d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-05T16:02:57.695Z</news:publication_date>
      <news:title>AgentS4D：代理完成任務不代表安全，逾六成執行同時觸發危險訊號</news:title>
    </news:news>
  </url>
</urlset>
