<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/era-回溯疫情預測的-11-優勢遭重估-資料修訂洩漏可解釋幾乎全部增益-eedf9a1c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T18:05:27.289Z</news:publication_date>
      <news:title>ERA 回溯疫情預測的 11% 優勢遭重估，資料修訂洩漏可解釋幾乎全部增益</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cloudflare-推出-kitesurf-以輕量-web-引擎取代-chromium-代理瀏覽資源用量降至約三分之一以下-edd4cc38</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T18:04:21.270Z</news:publication_date>
      <news:title>Cloudflare 推出 Kitesurf：以輕量 Web 引擎取代 Chromium，代理瀏覽資源用量降至約三分之一以下</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/herald-揭露搜尋代理可引用未檢索段落-單一成員檢查將實測攻擊率降至零-43b4aa38</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T16:04:28.071Z</news:publication_date>
      <news:title>HERALD 揭露搜尋代理可引用未檢索段落，單一成員檢查將實測攻擊率降至零</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/liquid-ai-開放-lfm2-5-2-6b-權重-2-69b-端側模型支援-128k-上下文與工具呼叫-f7a38588</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T16:03:21.742Z</news:publication_date>
      <news:title>Liquid AI 開放 LFM2.5-2.6B 權重，2.69B 端側模型支援 128K 上下文與工具呼叫</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cipo-以遮蔽檢索證據分配步驟獎勵-qwen2-5-7b-七項問答平均-f1-升至-0-504-95d8c521</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T14:03:39.660Z</news:publication_date>
      <news:title>CIPO 以遮蔽檢索證據分配步驟獎勵，Qwen2.5-7B 七項問答平均 F1 升至 0.504</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/合成臨床資料通過效用門檻仍可有-79-44-缺失-兩項修訂卻拉大來源分布差距-0eeee11d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T14:04:38.417Z</news:publication_date>
      <news:title>合成臨床資料通過效用門檻仍可有 79.44% 缺失，兩項修訂卻拉大來源分布差距</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/u-opsd-只靠模型內部投票訓練-qwen3-8b-五項數學平均提高-10-7-分-45319a10</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T12:03:13.505Z</news:publication_date>
      <news:title>u-OPSD 只靠模型內部投票訓練，Qwen3-8B 五項數學平均提高 10.7 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/bakron-將雙側-hessian-量化降至三次複雜度-8192-方陣核心較-yaqa-快-60-倍-5ea25158</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T12:04:02.715Z</news:publication_date>
      <news:title>BaKron 將雙側 Hessian 量化降至三次複雜度，8192 方陣核心較 YAQA 快 60 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/690-項代理技能實測-混合檢索-hit-5-達-73-5-加入-llm-知識圖譜反降-11-2-點-897db420</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T08:04:05.835Z</news:publication_date>
      <news:title>690 項代理技能實測：混合檢索 hit@5 達 73.5%，加入 LLM 知識圖譜反降 11.2 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hyper-es-先用梯度建立低維搜尋空間-數學推理平均高於-grpo-lora-1-ec6954f5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T10:03:26.435Z</news:publication_date>
      <news:title>Hyper-ES 先用梯度建立低維搜尋空間，數學推理平均高於 GRPO-LoRA 1%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/finevo-bench-以-120-項金融任務測代理跨任務演化-codex-配對增益-19-37-分-3a9453cc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T08:03:05.532Z</news:publication_date>
      <news:title>FinEvo-Bench 以 120 項金融任務測代理跨任務演化，Codex 配對增益 19.37 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/opera-以物理殘差約束實驗代理-無實質改善的加分決策由最高-39-0-降至-1-9-47c4a82f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T04:03:10.081Z</news:publication_date>
      <news:title>OPERA 以物理殘差約束實驗代理，無實質改善的加分決策由最高 39.0% 降至 1.9%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/uk-aisi-網攻測試出現-19-次越界行動-代理曾以假身分推動惡意-pr-7dc87e2f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T06:04:59.056Z</news:publication_date>
      <news:title>UK AISI 網攻測試出現 19 次越界行動，代理曾以假身分推動惡意 PR</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vag-在技能寫入前攔截代理污染-terminal-bench-子集達-72-pass-1-1d5360eb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T04:04:24.501Z</news:publication_date>
      <news:title>VaG 在技能寫入前攔截代理污染，Terminal-Bench 子集達 72% pass@1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/macro-重排凍結-transformer-層-六款模型平均準確率提高-5-0-個百分點-72841781</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T02:04:09.588Z</news:publication_date>
      <news:title>MACRO 重排凍結 Transformer 層，六款模型平均準確率提高 5.0 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/google-開放-weathernext-2-程式與權重-完整模型推論需-h100-級記憶體-2c9a300a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T02:03:03.519Z</news:publication_date>
      <news:title>Google 開放 WeatherNext 2 程式與權重，完整模型推論需 H100 級記憶體</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/witprobe-為四類注意力記憶建立執行期風險帳本-1-240-萬次讀取未超出預算-d0376f50</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T00:03:23.460Z</news:publication_date>
      <news:title>WitProbe 為四類注意力記憶建立執行期風險帳本，1,240 萬次讀取未超出預算</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/codegrep-將程式庫搜尋拆成-14b-專用代理-修復成功案例少用-19-token-ee2f10fe</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T00:02:40.361Z</news:publication_date>
      <news:title>CodeGrep 將程式庫搜尋拆成 14B 專用代理，修復成功案例少用 19% token</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/rrc-將生成式獎勵模型的排序轉成-grpo-訊號-alpacaeval-2-得分由-35-8-升至-41-3-b6a09499</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T22:03:10.151Z</news:publication_date>
      <news:title>RRC 將生成式獎勵模型的排序轉成 GRPO 訊號，AlpacaEval 2 得分由 35.8% 升至 41.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hope-從單眼影片預測手部壓力-opentouch-頂點接觸-f1-達-0-660-4ccca256</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T22:03:57.647Z</news:publication_date>
      <news:title>HOPE 從單眼影片預測手部壓力，OpenTouch 頂點接觸 F1 達 0.660</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/舊工具軌跡可劫持代理決策-contextpollute-bench-測得-qwen3-1-7b-有-32-1-正確答案被翻轉-3f11b052</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T20:02:55.613Z</news:publication_date>
      <news:title>舊工具軌跡可劫持代理決策：ContextPollute-Bench 測得 Qwen3-1.7B 有 32.1% 正確答案被翻轉</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/read-讓代理直接搜尋結構化長文件-51-題準確率達-58-8-但未顯著勝過-bm25-d5f6217c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T20:03:09.576Z</news:publication_date>
      <news:title>READ 讓代理直接搜尋結構化長文件，51 題準確率達 58.8%，但未顯著勝過 BM25</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cogvis-共用遙測影像變化先驗-十類查詢延遲降至-10-03-秒-98643c3f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T18:03:11.923Z</news:publication_date>
      <news:title>CogVis 共用遙測影像變化先驗，十類查詢延遲降至 10.03 秒</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/agentopsd-以遞迴信念分配代理獎勵-qwen2-5-7b-在-alfworld-達-89-1-a1795fab</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T18:01:47.895Z</news:publication_date>
      <news:title>AgentOPSD 以遞迴信念分配代理獎勵，Qwen2.5-7B 在 ALFWorld 達 89.1%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/appdeltaworld-以可執行-html-預測手機介面-mobilegym-成功率由-10-2-升至-14-1-cb7bee59</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T16:03:55.633Z</news:publication_date>
      <news:title>AppDeltaWorld 以可執行 HTML 預測手機介面，MobileGym 成功率由 10.2% 升至 14.1%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/gauge-以-1-560-次實體實驗檢驗模擬器與影片世界模型-沒有單一引擎全面勝出-ce41876f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T16:02:38.339Z</news:publication_date>
      <news:title>GAUGE 以 1,560 次實體實驗檢驗模擬器與影片世界模型，沒有單一引擎全面勝出</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/mist-以四種配對情境測試-選擇性信任-誤導訊息令-23-款模型平均掉-17-1-分-06e7fd0d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T14:02:49.637Z</news:publication_date>
      <news:title>MIST 以四種配對情境測試「選擇性信任」，誤導訊息令 23 款模型平均掉 17.1 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/程式化工具呼叫在-14-款模型中有-11-款不輸-json-長鏈任務差距達-18-8-個百分點-9fc48de7</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T14:03:52.237Z</news:publication_date>
      <news:title>程式化工具呼叫在 14 款模型中有 11 款不輸 JSON，長鏈任務差距達 18.8 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/geniworld-將機器人動作渲染成視覺條件-合成資料使實機成功率升至-69-0-1afe56be</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T12:03:40.656Z</news:publication_date>
      <news:title>GeniWorld 將機器人動作渲染成視覺條件，合成資料使實機成功率升至 69.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/calibforge-以求解器分歧校準終端任務-跨基準最高提升-30-04-個百分點-e98cc315</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T12:02:37.276Z</news:publication_date>
      <news:title>CalibForge 以求解器分歧校準終端任務，跨基準最高提升 30.04 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/2-190-段受控影片揭露-vlm-計數邊界-增加取樣幀仍無法保證事件軌跡正確-8c9aa159</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T10:03:44.060Z</news:publication_date>
      <news:title>2,190 段受控影片揭露 VLM 計數邊界：增加取樣幀仍無法保證事件軌跡正確</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/harnessopt-bench-將代理外殼最佳化變成評測-模型選擇影響約為-coding-harness-的-1-8-倍-5badd8e6</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T10:02:29.324Z</news:publication_date>
      <news:title>HarnessOpt-Bench 將代理外殼最佳化變成評測，模型選擇影響約為 coding harness 的 1.8 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/trajdebug-追蹤代理錯誤生命週期-失敗診斷讓重跑成功率平均提高-10-8-5618b18c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T08:02:36.276Z</news:publication_date>
      <news:title>TrajDebug 追蹤代理錯誤生命週期，失敗診斷讓重跑成功率平均提高 10.8%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/microevo-以-llm-與-mcts-搜尋處理器設計-pareto-品質最高提高-36-2-63c949bb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T08:03:42.233Z</news:publication_date>
      <news:title>MicroEvo 以 LLM 與 MCTS 搜尋處理器設計，Pareto 品質最高提高 36.2%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/padoc-以版面分支並行解析文件-單張-a800-吞吐量最高提高-118-f3a71df2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T06:03:50.115Z</news:publication_date>
      <news:title>PaDoc 以版面分支並行解析文件，單張 A800 吞吐量最高提高 118%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/視覺模型呼叫裁切工具不等於使用證據-六模型增益集中於少數有效軌跡-14b25b93</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T06:02:16.012Z</news:publication_date>
      <news:title>視覺模型呼叫裁切工具不等於使用證據，六模型增益集中於少數有效軌跡</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/envace-讓單一模型同時扮演代理與環境-三項工具評測綜合分升至-32-91-f6be87e8</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T04:02:13.763Z</news:publication_date>
      <news:title>EnvACE 讓單一模型同時扮演代理與環境，三項工具評測綜合分升至 32.91%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/新研究指出-adam-會打破矩陣分解對稱性-相同函數可收斂至不同注意力表示-768bd8b8</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T00:02:25.412Z</news:publication_date>
      <news:title>新研究指出 Adam 會打破矩陣分解對稱性，相同函數可收斂至不同注意力表示</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dash-依推理軌跡動態分配蒸餾權重-qwen3-三種規模皆優於固定權重基線-b1674689</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T02:02:22.195Z</news:publication_date>
      <news:title>DASH 依推理軌跡動態分配蒸餾權重，Qwen3 三種規模皆優於固定權重基線</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/reasoning-core-以-50-種程序生成器擴充監督資料-3b-模型-drop-f1-由-33-1-升至-41-7-edf95972</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T00:02:18.768Z</news:publication_date>
      <news:title>Reasoning Core 以 50 種程序生成器擴充監督資料，3B 模型 DROP F1 由 33.1 升至 41.7</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/36-710-個-github-專案只留下-85-份代理計畫-多數尚非長期維護文件-0bfbc5ed</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T22:05:07.244Z</news:publication_date>
      <news:title>36,710 個 GitHub 專案只留下 85 份代理計畫，多數尚非長期維護文件</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/meta-推出-muse-code-以持久背景代理與事件日誌支撐長時程程式任務-0dce067c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T22:03:54.872Z</news:publication_date>
      <news:title>Meta 推出 Muse Code：以持久背景代理與事件日誌支撐長時程程式任務</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/canary-tools-將代理選錯工具拆成六類-八款模型受騙率相差約-36-倍-f68fe4af</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T20:02:57.709Z</news:publication_date>
      <news:title>Canary Tools 將代理選錯工具拆成六類，八款模型受騙率相差約 36 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/argus-讓固定權重代理累積經驗-swe-bench-pro-準確率由-59-升至約-78-7d22f77a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-06T20:01:58.602Z</news:publication_date>
      <news:title>Argus 讓固定權重代理累積經驗，SWE-Bench Pro 準確率由 59% 升至約 78%</news:title>
    </news:news>
  </url>
</urlset>
