<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/harnessopt-bench-測試模型改寫代理外殼-模型選擇的影響約為編碼工具-1-8-倍-76ad33a2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T18:01:56.310Z</news:publication_date>
      <news:title>HarnessOpt-Bench 測試模型改寫代理外殼，模型選擇的影響約為編碼工具 1.8 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/影片模型在高頻多事件區僅答對-0-2-增加取樣幀仍未找回事件證據-e1858030</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T16:02:42.951Z</news:publication_date>
      <news:title>影片模型在高頻多事件區僅答對 0.2%，增加取樣幀仍未找回事件證據</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dash-依推理分歧動態分配蒸餾權重-qwen3-1-7b-數學平均提高-3-2-分-e345f9c0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T00:02:14.160Z</news:publication_date>
      <news:title>DASH 依推理分歧動態分配蒸餾權重，Qwen3-1.7B 數學平均提高 3.2 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/iarcs-讓-llm-生成-3d-場景獎勵程式-物件碰撞率由-52-67-降至-40-45-8e8724ba</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T00:02:59.008Z</news:publication_date>
      <news:title>iARCS 讓 LLM 生成 3D 場景獎勵程式，物件碰撞率由 52.67% 降至 40.45%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/tutormoments-重播真實教學決策-測出七款-llm-預設傾向過度提示-225a2f55</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T02:03:29.225Z</news:publication_date>
      <news:title>TutorMoments 重播真實教學決策，測出七款 LLM 預設傾向過度提示</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/otter-以棋步歷史與時限預測人類決策-15-3m-參數模型報告-55-23-top-1-8e299bdd</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T22:03:49.787Z</news:publication_date>
      <news:title>Otter 以棋步歷史與時限預測人類決策，15.3M 參數模型報告 55.23% top-1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/smrc-sd-先核對代理狀態再蒸餾成功軌跡-qwen3-1-7b-任務成功率提高約-12-個百分點-cd056bef</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T22:02:40.250Z</news:publication_date>
      <news:title>SMRC-SD 先核對代理狀態再蒸餾成功軌跡，Qwen3-1.7B 任務成功率提高約 12 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/era-回溯疫情預測的-11-優勢遭重估-資料修訂洩漏可解釋幾乎全部增益-eedf9a1c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T18:05:27.289Z</news:publication_date>
      <news:title>ERA 回溯疫情預測的 11% 優勢遭重估，資料修訂洩漏可解釋幾乎全部增益</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cloudflare-推出-kitesurf-以輕量-web-引擎取代-chromium-代理瀏覽資源用量降至約三分之一以下-edd4cc38</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T18:04:21.270Z</news:publication_date>
      <news:title>Cloudflare 推出 Kitesurf：以輕量 Web 引擎取代 Chromium，代理瀏覽資源用量降至約三分之一以下</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/herald-揭露搜尋代理可引用未檢索段落-單一成員檢查將實測攻擊率降至零-43b4aa38</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T16:04:28.071Z</news:publication_date>
      <news:title>HERALD 揭露搜尋代理可引用未檢索段落，單一成員檢查將實測攻擊率降至零</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/liquid-ai-開放-lfm2-5-2-6b-權重-2-69b-端側模型支援-128k-上下文與工具呼叫-f7a38588</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T16:03:21.742Z</news:publication_date>
      <news:title>Liquid AI 開放 LFM2.5-2.6B 權重，2.69B 端側模型支援 128K 上下文與工具呼叫</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cipo-以遮蔽檢索證據分配步驟獎勵-qwen2-5-7b-七項問答平均-f1-升至-0-504-95d8c521</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T14:03:39.660Z</news:publication_date>
      <news:title>CIPO 以遮蔽檢索證據分配步驟獎勵，Qwen2.5-7B 七項問答平均 F1 升至 0.504</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/合成臨床資料通過效用門檻仍可有-79-44-缺失-兩項修訂卻拉大來源分布差距-0eeee11d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T14:04:38.417Z</news:publication_date>
      <news:title>合成臨床資料通過效用門檻仍可有 79.44% 缺失，兩項修訂卻拉大來源分布差距</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/u-opsd-只靠模型內部投票訓練-qwen3-8b-五項數學平均提高-10-7-分-45319a10</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T12:03:13.505Z</news:publication_date>
      <news:title>u-OPSD 只靠模型內部投票訓練，Qwen3-8B 五項數學平均提高 10.7 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/bakron-將雙側-hessian-量化降至三次複雜度-8192-方陣核心較-yaqa-快-60-倍-5ea25158</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T12:04:02.715Z</news:publication_date>
      <news:title>BaKron 將雙側 Hessian 量化降至三次複雜度，8192 方陣核心較 YAQA 快 60 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/690-項代理技能實測-混合檢索-hit-5-達-73-5-加入-llm-知識圖譜反降-11-2-點-897db420</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T08:04:05.835Z</news:publication_date>
      <news:title>690 項代理技能實測：混合檢索 hit@5 達 73.5%，加入 LLM 知識圖譜反降 11.2 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hyper-es-先用梯度建立低維搜尋空間-數學推理平均高於-grpo-lora-1-ec6954f5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T10:03:26.435Z</news:publication_date>
      <news:title>Hyper-ES 先用梯度建立低維搜尋空間，數學推理平均高於 GRPO-LoRA 1%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/finevo-bench-以-120-項金融任務測代理跨任務演化-codex-配對增益-19-37-分-3a9453cc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T08:03:05.532Z</news:publication_date>
      <news:title>FinEvo-Bench 以 120 項金融任務測代理跨任務演化，Codex 配對增益 19.37 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/opera-以物理殘差約束實驗代理-無實質改善的加分決策由最高-39-0-降至-1-9-47c4a82f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T04:03:10.081Z</news:publication_date>
      <news:title>OPERA 以物理殘差約束實驗代理，無實質改善的加分決策由最高 39.0% 降至 1.9%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/uk-aisi-網攻測試出現-19-次越界行動-代理曾以假身分推動惡意-pr-7dc87e2f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T06:04:59.056Z</news:publication_date>
      <news:title>UK AISI 網攻測試出現 19 次越界行動，代理曾以假身分推動惡意 PR</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vag-在技能寫入前攔截代理污染-terminal-bench-子集達-72-pass-1-1d5360eb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T04:04:24.501Z</news:publication_date>
      <news:title>VaG 在技能寫入前攔截代理污染，Terminal-Bench 子集達 72% pass@1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/macro-重排凍結-transformer-層-六款模型平均準確率提高-5-0-個百分點-72841781</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T02:04:09.588Z</news:publication_date>
      <news:title>MACRO 重排凍結 Transformer 層，六款模型平均準確率提高 5.0 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/google-開放-weathernext-2-程式與權重-完整模型推論需-h100-級記憶體-2c9a300a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T02:03:03.519Z</news:publication_date>
      <news:title>Google 開放 WeatherNext 2 程式與權重，完整模型推論需 H100 級記憶體</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/witprobe-為四類注意力記憶建立執行期風險帳本-1-240-萬次讀取未超出預算-d0376f50</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T00:03:23.460Z</news:publication_date>
      <news:title>WitProbe 為四類注意力記憶建立執行期風險帳本，1,240 萬次讀取未超出預算</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/codegrep-將程式庫搜尋拆成-14b-專用代理-修復成功案例少用-19-token-ee2f10fe</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T00:02:40.361Z</news:publication_date>
      <news:title>CodeGrep 將程式庫搜尋拆成 14B 專用代理，修復成功案例少用 19% token</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/rrc-將生成式獎勵模型的排序轉成-grpo-訊號-alpacaeval-2-得分由-35-8-升至-41-3-b6a09499</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T22:03:10.151Z</news:publication_date>
      <news:title>RRC 將生成式獎勵模型的排序轉成 GRPO 訊號，AlpacaEval 2 得分由 35.8% 升至 41.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hope-從單眼影片預測手部壓力-opentouch-頂點接觸-f1-達-0-660-4ccca256</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T22:03:57.647Z</news:publication_date>
      <news:title>HOPE 從單眼影片預測手部壓力，OpenTouch 頂點接觸 F1 達 0.660</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/舊工具軌跡可劫持代理決策-contextpollute-bench-測得-qwen3-1-7b-有-32-1-正確答案被翻轉-3f11b052</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T20:02:55.613Z</news:publication_date>
      <news:title>舊工具軌跡可劫持代理決策：ContextPollute-Bench 測得 Qwen3-1.7B 有 32.1% 正確答案被翻轉</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/read-讓代理直接搜尋結構化長文件-51-題準確率達-58-8-但未顯著勝過-bm25-d5f6217c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-07T20:03:09.576Z</news:publication_date>
      <news:title>READ 讓代理直接搜尋結構化長文件，51 題準確率達 58.8%，但未顯著勝過 BM25</news:title>
    </news:news>
  </url>
</urlset>
