<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/tempo-依實測執行時間分派-moe-專家-qwen3-235b-的-p99-token-延遲降低約-15-6-b58b2472</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T16:04:05.233Z</news:publication_date>
      <news:title>TEMPO 依實測執行時間分派 MoE 專家，Qwen3‑235B 的 p99 token 延遲降低約 15.6%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vero-把程式代理推至整個-lean-4-儲存庫-最強設定只完整解出-27-43-題-1d51347d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T16:02:40.041Z</news:publication_date>
      <news:title>Vero 把程式代理推至整個 Lean 4 儲存庫，最強設定只完整解出 27／43 題</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/omniscientist-讓多模態證據貫穿研究流程-直接感知版本在配對評審中勝出-85-e4f5b6c1</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T14:03:01.271Z</news:publication_date>
      <news:title>OmniScientist 讓多模態證據貫穿研究流程，直接感知版本在配對評審中勝出 85%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/statebridge-以閉式對齊傳送代理隱藏狀態-26-組測試有-22-組最佳或並列最佳-2393755a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T14:01:57.018Z</news:publication_date>
      <news:title>StateBridge 以閉式對齊傳送代理隱藏狀態，26 組測試有 22 組最佳或並列最佳</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/技能演化會保存惡意捷徑-21-組代理設定全部寫出不安全技能-cd348be6</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T12:02:49.820Z</news:publication_date>
      <news:title>技能演化會保存惡意捷徑：21 組代理設定全部寫出不安全技能</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/flashdrive-聯合快取-推測解碼與-w4a8-將-10b-自駕-vla-延遲壓至-151-毫秒-66012976</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T12:02:00.782Z</news:publication_date>
      <news:title>FlashDrive 聯合快取、推測解碼與 W4A8，將 10B 自駕 VLA 延遲壓至 151 毫秒</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/alayaworld-v1-1-改用串流-3d-點快取-長時影片一致性評分升至-89-5-c2fcd872</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T10:02:29.140Z</news:publication_date>
      <news:title>AlayaWorld v1.1 改用串流 3D 點快取，長時影片一致性評分升至 89.5</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/gcache-依全程誤差影響配置擴散快取-wan2-1-同速下-lpips-降至-0-0316-99eab0dd</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T10:03:24.134Z</news:publication_date>
      <news:title>GCache 依全程誤差影響配置擴散快取，Wan2.1 同速下 LPIPS 降至 0.0316</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/quotebench-固定模型輸出重播命令路徑-單層-shell-解析可令成功率跌逾-70-點-c7828756</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T08:03:34.336Z</news:publication_date>
      <news:title>QuoteBench 固定模型輸出重播命令路徑，單層 shell 解析可令成功率跌逾 70 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vtoken-以-token-級虛擬化回收-kv-cache-單張-h100-吞吐最高提高-37-2b3d6e0f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T08:02:19.782Z</news:publication_date>
      <news:title>vToken 以 token 級虛擬化回收 KV cache，單張 H100 吞吐最高提高 37%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/lazytrain-用混合整數排程重疊記憶體搬移-單張-h800-訓練-27b-模型快-1-24-倍-e86e8d54</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T02:03:26.815Z</news:publication_date>
      <news:title>LazyTrain 用混合整數排程重疊記憶體搬移，單張 H800 訓練 27B 模型快 1.24 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/trie-automata-預算-token-mask-vllm-有限選項解碼吞吐達-xgrammar-的-29-倍-6b82d030</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T04:02:22.288Z</news:publication_date>
      <news:title>Trie Automata 預算 token mask，vLLM 有限選項解碼吞吐達 XGrammar 的 29 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/長上下文訓練出現倒-u-曲線-窗口超過任務長度約-16-至-32-倍後能力轉弱-eb9b2611</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T02:02:34.990Z</news:publication_date>
      <news:title>長上下文訓練出現倒 U 曲線：窗口超過任務長度約 16 至 32 倍後能力轉弱</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/混合線性注意力在完整注意力層前形成巨量啟用尖峰-規律橫跨-1-2b-至-397b-模型-fab4ab86</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T00:10:40.176Z</news:publication_date>
      <news:title>混合線性注意力在完整注意力層前形成巨量啟用尖峰，規律橫跨 1.2B 至 397B 模型</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/三套代理基準的模型主效應低於-3-ddr-將排行榜改寫成部署可靠度問題-aba4ec2a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-14T00:09:27.680Z</news:publication_date>
      <news:title>三套代理基準的模型主效應低於 3%，DDR 將排行榜改寫成部署可靠度問題</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/llm-代理把-hpc-工作送上異質叢集-描述性硬體資料令成功率由-48-升至-87-9084dddc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T22:03:30.990Z</news:publication_date>
      <news:title>LLM 代理把 HPC 工作送上異質叢集，描述性硬體資料令成功率由 48% 升至 87%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/gsr-將評分規則編譯成型別圖-llm-裁判精確分數一致率最高增-6-75-點-08d92872</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T22:02:14.012Z</news:publication_date>
      <news:title>GSR 將評分規則編譯成型別圖，LLM 裁判精確分數一致率最高增 6.75 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/64-路自一致投票在多數-gpqa-難題上反而降低單題準確率-c89cf6bb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T20:03:20.148Z</news:publication_date>
      <news:title>64 路自一致投票在多數 GPQA 難題上反而降低單題準確率</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/minimax-h3-公開-33b-基礎權重-原生同步生成影片與立體聲-75f3a8af</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T20:02:21.108Z</news:publication_date>
      <news:title>MiniMax H3 公開 33B 基礎權重，原生同步生成影片與立體聲</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/harness-if-分離-本來就會做-與真正遵令-12-款程式代理落差最高-7-4-點-930e57dc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T18:02:50.375Z</news:publication_date>
      <news:title>Harness‑IF 分離「本來就會做」與真正遵令，12 款程式代理落差最高 7.4 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vicbench-追溯100個漏洞的首次引入提交-現有自動方法最高僅40-1-f1-18cef9c0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T18:03:36.590Z</news:publication_date>
      <news:title>VICBench 追溯100個漏洞的首次引入提交，現有自動方法最高僅40.1% F1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ltx-2-5-以擴散解碼器與單次多鏡生成更新開放影音模型-e8e571be</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T16:03:58.226Z</news:publication_date>
      <news:title>LTX‑2.5 以擴散解碼器與單次多鏡生成更新開放影音模型</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/qwen3-8-首次開放-2-4t-參數-max-級權重-每個-token-啟用-95b-e2ccf97c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T16:02:39.029Z</news:publication_date>
      <news:title>Qwen3.8 首次開放 2.4T 參數 Max 級權重，每個 token 啟用 95B</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/相同模型排名隨-token-上限反轉-四模型測試錄得-56-476-次推論-00ad3e4d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T12:03:24.595Z</news:publication_date>
      <news:title>相同模型排名隨 token 上限反轉，四模型測試錄得 56,476 次推論</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/bench2robust-注入九類工具故障-69-70-組代理測試出現性能倒退-030ad05d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T12:02:23.490Z</news:publication_date>
      <news:title>Bench2Robust 注入九類工具故障，69／70 組代理測試出現性能倒退</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/claude-新模型將文字水印寫入生成過程-api-與-claude-code-亦無例外-0479d357</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T10:03:05.324Z</news:publication_date>
      <news:title>Claude 新模型將文字水印寫入生成過程，API 與 Claude Code 亦無例外</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/stateflow-以持久-3d-狀態取代逐鏡重生成-影片預視使用者評分達-4-5-5-f4a8f716</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T10:04:02.593Z</news:publication_date>
      <news:title>StateFlow 以持久 3D 狀態取代逐鏡重生成，影片預視使用者評分達 4.5／5</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/強模型把任務規則編譯成-harness-gpt-5-4-mini-準確率由-0-488-升至-0-912-64eb15fd</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T08:04:58.715Z</news:publication_date>
      <news:title>強模型把任務規則編譯成 harness，GPT‑5.4‑mini 準確率由 0.488 升至 0.912</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/lfm2-5-vl-3b-以約-3gb-記憶體執行視覺模型-原生支援-vllm-llama-cpp-與-mlx-61fb095e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T08:03:31.183Z</news:publication_date>
      <news:title>LFM2.5‑VL‑3B 以約 3GB 記憶體執行視覺模型，原生支援 vLLM、llama.cpp 與 MLX</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/長期記憶摘要只保留-3-05-時間表達-一句提示令時序問答準確率提高-31-4-點-bf54244f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T06:03:16.716Z</news:publication_date>
      <news:title>長期記憶摘要只保留 3.05% 時間表達，一句提示令時序問答準確率提高 31.4 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ctbench-用-234-個電信故障沙箱重測代理-最佳根因分析正確率僅-47-62-18e8c120</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T06:01:55.793Z</news:publication_date>
      <news:title>CTBench 用 234 個電信故障沙箱重測代理，最佳根因分析正確率僅 47.62%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/oeis-open-用-lean-驗證-492-個未解猜想-但交叉檢查把通過數由-147-修正為-144-fbe5b845</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T04:03:37.316Z</news:publication_date>
      <news:title>OEIS Open 用 Lean 驗證 492 個未解猜想，但交叉檢查把通過數由 147 修正為 144</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/vakra-重播逾-8-000-個本機-api-代理遇多跳與政策限制時準確率驟降-dc6a5789</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T04:02:38.345Z</news:publication_date>
      <news:title>VAKRA 重播逾 8,000 個本機 API，代理遇多跳與政策限制時準確率驟降</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/相關技能也可能拖垮代理-307-個失敗案例多源於錯誤實作與過度驗證-c1131f4d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T02:03:04.698Z</news:publication_date>
      <news:title>相關技能也可能拖垮代理：307 個失敗案例多源於錯誤實作與過度驗證</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ling-3-0-tiny-每個-token-只啟用-1-3b-參數-但本機執行仍依賴專用-runtime-分支-25729bc0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T02:03:57.413Z</news:publication_date>
      <news:title>Ling‑3.0‑tiny 每個 token 只啟用 1.3B 參數，但本機執行仍依賴專用 runtime 分支</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dasharena-要求模型附上可重播操作軌跡-揭露儀表板-能顯示但不能用-的落差-4c5bd8cb</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T00:02:42.600Z</news:publication_date>
      <news:title>DashArena 要求模型附上可重播操作軌跡，揭露儀表板「能顯示但不能用」的落差</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/io-factory-以十萬個模擬角色重播-ai-影響行動-但不能推算真實說服效果-ad5ef3dc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-13T00:03:35.178Z</news:publication_date>
      <news:title>IO Factory 以十萬個模擬角色重播 AI 影響行動，但不能推算真實說服效果</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/retree-在證據衝突時回滾搜尋樹-qwen3-8b-整體答案準確率由-30-1-升至-44-0-2b3567d0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T22:04:11.854Z</news:publication_date>
      <news:title>ReTree 在證據衝突時回滾搜尋樹，Qwen3‑8B 整體答案準確率由 30.1%升至 44.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/thinkretrieve-在推理途中檢索解題範例-qwen3-1-7b-的-aime-2025-準確率提高-13-4-點-fc8d3dba</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T22:02:54.398Z</news:publication_date>
      <news:title>ThinkRetrieve 在推理途中檢索解題範例，Qwen3‑1.7B 的 AIME 2025 準確率提高 13.4 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/加入相容形容詞仍令-sae-遺失-20-至-60-啟用特徵-挑戰-特徵袋-直覺-beb74ea2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T20:03:30.133Z</news:publication_date>
      <news:title>加入相容形容詞仍令 SAE 遺失 20% 至 60% 啟用特徵，挑戰「特徵袋」直覺</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/skiller-把-skill-md-當成可優化策略-qwen3-5-9b-在軟體工程測試達-82-8-92c44f87</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T20:02:41.821Z</news:publication_date>
      <news:title>SKILLER 把 SKILL.md 當成可優化策略，Qwen3.5‑9B 在軟體工程測試達 82.8%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/muse-glimmer-30b-以-17gb-量化權重與-dflash-草稿模型-把多模態代理搬上單張顯卡-a566f334</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T18:03:14.871Z</news:publication_date>
      <news:title>Muse Glimmer 30B 以 17GB 量化權重與 DFlash 草稿模型，把多模態代理搬上單張顯卡</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/雙代理研究系統從反覆失敗中導出-grothendieck-常數新下界-但關鍵轉向仍由人類提出-ca12597d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-12T18:04:15.828Z</news:publication_date>
      <news:title>雙代理研究系統從反覆失敗中導出 Grothendieck 常數新下界，但關鍵轉向仍由人類提出</news:title>
    </news:news>
  </url>
</urlset>
