<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/64-路自一致投票在多數-gpqa-難題上反而降低單題準確率-c89cf6bb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T20:03:20.148Z</news:publication_date>
      <news:title>64 路自一致投票在多數 GPQA 難題上反而降低單題準確率</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/minimax-h3-公開-33b-基礎權重-原生同步生成影片與立體聲-75f3a8af</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T20:02:21.108Z</news:publication_date>
      <news:title>MiniMax H3 公開 33B 基礎權重，原生同步生成影片與立體聲</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/harness-if-分離-本來就會做-與真正遵令-12-款程式代理落差最高-7-4-點-930e57dc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T18:02:50.375Z</news:publication_date>
      <news:title>Harness‑IF 分離「本來就會做」與真正遵令，12 款程式代理落差最高 7.4 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vicbench-追溯100個漏洞的首次引入提交-現有自動方法最高僅40-1-f1-18cef9c0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T18:03:36.590Z</news:publication_date>
      <news:title>VICBench 追溯100個漏洞的首次引入提交，現有自動方法最高僅40.1% F1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ltx-2-5-以擴散解碼器與單次多鏡生成更新開放影音模型-e8e571be</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T16:03:58.226Z</news:publication_date>
      <news:title>LTX‑2.5 以擴散解碼器與單次多鏡生成更新開放影音模型</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/qwen3-8-首次開放-2-4t-參數-max-級權重-每個-token-啟用-95b-e2ccf97c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T16:02:39.029Z</news:publication_date>
      <news:title>Qwen3.8 首次開放 2.4T 參數 Max 級權重，每個 token 啟用 95B</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/相同模型排名隨-token-上限反轉-四模型測試錄得-56-476-次推論-00ad3e4d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T12:03:24.595Z</news:publication_date>
      <news:title>相同模型排名隨 token 上限反轉，四模型測試錄得 56,476 次推論</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/bench2robust-注入九類工具故障-69-70-組代理測試出現性能倒退-030ad05d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T12:02:23.490Z</news:publication_date>
      <news:title>Bench2Robust 注入九類工具故障，69／70 組代理測試出現性能倒退</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/claude-新模型將文字水印寫入生成過程-api-與-claude-code-亦無例外-0479d357</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T10:03:05.324Z</news:publication_date>
      <news:title>Claude 新模型將文字水印寫入生成過程，API 與 Claude Code 亦無例外</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/stateflow-以持久-3d-狀態取代逐鏡重生成-影片預視使用者評分達-4-5-5-f4a8f716</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T10:04:02.593Z</news:publication_date>
      <news:title>StateFlow 以持久 3D 狀態取代逐鏡重生成，影片預視使用者評分達 4.5／5</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/強模型把任務規則編譯成-harness-gpt-5-4-mini-準確率由-0-488-升至-0-912-64eb15fd</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T08:04:58.715Z</news:publication_date>
      <news:title>強模型把任務規則編譯成 harness，GPT‑5.4‑mini 準確率由 0.488 升至 0.912</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/lfm2-5-vl-3b-以約-3gb-記憶體執行視覺模型-原生支援-vllm-llama-cpp-與-mlx-61fb095e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T08:03:31.183Z</news:publication_date>
      <news:title>LFM2.5‑VL‑3B 以約 3GB 記憶體執行視覺模型，原生支援 vLLM、llama.cpp 與 MLX</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/長期記憶摘要只保留-3-05-時間表達-一句提示令時序問答準確率提高-31-4-點-bf54244f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T06:03:16.716Z</news:publication_date>
      <news:title>長期記憶摘要只保留 3.05% 時間表達，一句提示令時序問答準確率提高 31.4 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ctbench-用-234-個電信故障沙箱重測代理-最佳根因分析正確率僅-47-62-18e8c120</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T06:01:55.793Z</news:publication_date>
      <news:title>CTBench 用 234 個電信故障沙箱重測代理，最佳根因分析正確率僅 47.62%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/oeis-open-用-lean-驗證-492-個未解猜想-但交叉檢查把通過數由-147-修正為-144-fbe5b845</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T04:03:37.316Z</news:publication_date>
      <news:title>OEIS Open 用 Lean 驗證 492 個未解猜想，但交叉檢查把通過數由 147 修正為 144</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vakra-重播逾-8-000-個本機-api-代理遇多跳與政策限制時準確率驟降-dc6a5789</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T04:02:38.345Z</news:publication_date>
      <news:title>VAKRA 重播逾 8,000 個本機 API，代理遇多跳與政策限制時準確率驟降</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/相關技能也可能拖垮代理-307-個失敗案例多源於錯誤實作與過度驗證-c1131f4d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T02:03:04.698Z</news:publication_date>
      <news:title>相關技能也可能拖垮代理：307 個失敗案例多源於錯誤實作與過度驗證</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ling-3-0-tiny-每個-token-只啟用-1-3b-參數-但本機執行仍依賴專用-runtime-分支-25729bc0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T02:03:57.413Z</news:publication_date>
      <news:title>Ling‑3.0‑tiny 每個 token 只啟用 1.3B 參數，但本機執行仍依賴專用 runtime 分支</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dasharena-要求模型附上可重播操作軌跡-揭露儀表板-能顯示但不能用-的落差-4c5bd8cb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T00:02:42.600Z</news:publication_date>
      <news:title>DashArena 要求模型附上可重播操作軌跡，揭露儀表板「能顯示但不能用」的落差</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/io-factory-以十萬個模擬角色重播-ai-影響行動-但不能推算真實說服效果-ad5ef3dc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T00:03:35.178Z</news:publication_date>
      <news:title>IO Factory 以十萬個模擬角色重播 AI 影響行動，但不能推算真實說服效果</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/retree-在證據衝突時回滾搜尋樹-qwen3-8b-整體答案準確率由-30-1-升至-44-0-2b3567d0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T22:04:11.854Z</news:publication_date>
      <news:title>ReTree 在證據衝突時回滾搜尋樹，Qwen3‑8B 整體答案準確率由 30.1%升至 44.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/thinkretrieve-在推理途中檢索解題範例-qwen3-1-7b-的-aime-2025-準確率提高-13-4-點-fc8d3dba</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T22:02:54.398Z</news:publication_date>
      <news:title>ThinkRetrieve 在推理途中檢索解題範例，Qwen3‑1.7B 的 AIME 2025 準確率提高 13.4 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/加入相容形容詞仍令-sae-遺失-20-至-60-啟用特徵-挑戰-特徵袋-直覺-beb74ea2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T20:03:30.133Z</news:publication_date>
      <news:title>加入相容形容詞仍令 SAE 遺失 20% 至 60% 啟用特徵，挑戰「特徵袋」直覺</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/skiller-把-skill-md-當成可優化策略-qwen3-5-9b-在軟體工程測試達-82-8-92c44f87</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T20:02:41.821Z</news:publication_date>
      <news:title>SKILLER 把 SKILL.md 當成可優化策略，Qwen3.5‑9B 在軟體工程測試達 82.8%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/muse-glimmer-30b-以-17gb-量化權重與-dflash-草稿模型-把多模態代理搬上單張顯卡-a566f334</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T18:03:14.871Z</news:publication_date>
      <news:title>Muse Glimmer 30B 以 17GB 量化權重與 DFlash 草稿模型，把多模態代理搬上單張顯卡</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/雙代理研究系統從反覆失敗中導出-grothendieck-常數新下界-但關鍵轉向仍由人類提出-ca12597d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T18:04:15.828Z</news:publication_date>
      <news:title>雙代理研究系統從反覆失敗中導出 Grothendieck 常數新下界，但關鍵轉向仍由人類提出</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/chemworld-將化學流程編譯成可重播代理環境-失敗操作也保留交易收據-0664ee73</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T16:04:20.374Z</news:publication_date>
      <news:title>ChemWorld 將化學流程編譯成可重播代理環境，失敗操作也保留交易收據</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/evomem-將跨任務優化經驗寫成持久記憶-演化搜尋平均加速-5-93-倍-d5c81318</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T16:03:11.706Z</news:publication_date>
      <news:title>EvoMem 將跨任務優化經驗寫成持久記憶，演化搜尋平均加速 5.93 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/acm-將三種代理框架投影成版本化設定圖-統一追蹤變更影響與執行來源-81b98852</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T14:03:22.833Z</news:publication_date>
      <news:title>ACM 將三種代理框架投影成版本化設定圖，統一追蹤變更影響與執行來源</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/misa-t-以-kv-駐留時間調度混合式-rl-rollout-step3-7-吞吐提高-53-3-e3acc0df</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T14:02:34.343Z</news:publication_date>
      <news:title>MISA-T 以 KV 駐留時間調度混合式 RL rollout，Step3.7 吞吐提高 53.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/unif-moe-先抽共享區塊再路由殘餘計算-較-top-2-gmoe-降低-45-2-推論時間-4b1d5dca</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T12:04:35.205Z</news:publication_date>
      <news:title>UniF-MoE 先抽共享區塊再路由殘餘計算，較 top-2 GMoE 降低 45.2% 推論時間</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/跨語言提示改變代理工具路徑-四款大型模型僅保留-71-至-73-行動策略-b0a81e65</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T12:03:15.932Z</news:publication_date>
      <news:title>跨語言提示改變代理工具路徑，四款大型模型僅保留 71% 至 73% 行動策略</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/reround-用擴散先驗重選量化捨入-3-4-位元小模型平均準確率最高增加-1-6-點-3061c9af</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T10:04:35.143Z</news:publication_date>
      <news:title>ReRound 用擴散先驗重選量化捨入，3／4 位元小模型平均準確率最高增加 1.6 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/全域除錯記憶讓-autoresearch-代理避免重踩錯誤-aide-金牌執行由-22-次增至-38-次-8717522c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T10:03:34.440Z</news:publication_date>
      <news:title>全域除錯記憶讓 AutoResearch 代理避免重踩錯誤，AIDE 金牌執行由 22 次增至 38 次</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/gpt-oss-20b-檢查重構-diff-找出-rope-13-類缺陷-12-項獲維護者接受-19b48998</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T08:06:43.910Z</news:publication_date>
      <news:title>gpt-oss-20b 檢查重構 diff 找出 Rope 13 類缺陷，12 項獲維護者接受</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/xcot-vla-用-2-至-6-個動作-token-壓縮自駕推理-h100-介面延遲最低-38-6-毫秒-0574957a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T08:05:11.318Z</news:publication_date>
      <news:title>XCoT-VLA 用 2 至 6 個動作 token 壓縮自駕推理，H100 介面延遲最低 38.6 毫秒</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/redagentbench-以服務最終狀態重算代理攻擊率-文字軌跡漏判最高-11-72-個百分點-deb46abe</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T04:03:07.481Z</news:publication_date>
      <news:title>REDAgentBench 以服務最終狀態重算代理攻擊率，文字軌跡漏判最高 11.72 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/代理指令檔平均成長-226-研究以-規則理由-抑制-claude-md-膨脹-f3d5f03e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T04:02:14.957Z</news:publication_date>
      <news:title>代理指令檔平均成長 226%，研究以「規則理由」抑制 CLAUDE.md 膨脹</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/onedrive-photos-誤推至企業電腦後-微軟補上獨立移除與登入管制-83d5fbb2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T02:03:10.438Z</news:publication_date>
      <news:title>OneDrive Photos 誤推至企業電腦後，微軟補上獨立移除與登入管制</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/mevis-audio-冠軍方案以候選遮罩共識取代單一模型判斷-d6f6039f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T02:04:12.398Z</news:publication_date>
      <news:title>MeViS-Audio 冠軍方案以候選遮罩共識取代單一模型判斷</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/oeo-讓-gpt-5-5-自行編排代理改進流程-14-組比較取得-12-勝-a826867d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T00:05:11.687Z</news:publication_date>
      <news:title>OEO 讓 GPT‑5.5 自行編排代理改進流程，14 組比較取得 12 勝</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/llm-漏洞驗證產物重跑後-30-個案例僅-10-個通過修補版本反證-4ae70cad</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T00:03:56.066Z</news:publication_date>
      <news:title>LLM 漏洞驗證產物重跑後，30 個案例僅 10 個通過修補版本反證</news:title>
    </news:news>
  </url>
</urlset>
