<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/cellworld-改以潛在細胞表徵預訓練-5-74m-模型在-18-項空間轉錄體任務勝過所列基線-a17dc5c4</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T06:03:48.608Z</news:publication_date>
      <news:title>CellWorld 改以潛在細胞表徵預訓練，5.74M 模型在 18 項空間轉錄體任務勝過所列基線</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/agentpatch-免訓練修補多模態代理合併-六項基準平均分由-54-5-升至-56-6-0f6fce76</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T06:02:36.585Z</news:publication_date>
      <news:title>AgentPatch 免訓練修補多模態代理合併，六項基準平均分由 54.5 升至 56.6</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/open-science-v0-12-1-將科研代理的壓縮-執行與產物證據留在本機工作區-73248d22</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T04:03:31.139Z</news:publication_date>
      <news:title>Open Science v0.12.1 將科研代理的壓縮、執行與產物證據留在本機工作區</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/openai-暫停部分-astra-開發-初測無法排除模型已達-critical-網攻能力-26d5d557</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T04:02:43.008Z</news:publication_date>
      <news:title>OpenAI 暫停部分 Astra 開發：初測無法排除模型已達「Critical」網攻能力</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/didpo-把程式-diff-拆成信用單元-qwen2-5-coder-7b-主要評測平均升至-48-4-dd2b8db5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T02:04:45.548Z</news:publication_date>
      <news:title>DiDPO 把程式 diff 拆成信用單元，Qwen2.5-Coder-7B 主要評測平均升至 48.4%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/fisher-r1-以可驗證-p-值訓練統計代理-p-hard-嚴格-pass-1-達-33-0-8926a0a0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T02:03:21.063Z</news:publication_date>
      <news:title>Fisher-R1 以可驗證 p 值訓練統計代理，P-Hard 嚴格 pass@1 達 33.0%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ling-3-0-flash-int4-單機實測達-38-7-token-s-錯用主線-vllm-可能無聲產生錯誤輸出-cda07d6e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T00:04:10.660Z</news:publication_date>
      <news:title>Ling-3.0-flash INT4 單機實測達 38.7 token/s，錯用主線 vLLM 可能無聲產生錯誤輸出</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/minimax-開放-h3-基礎權重-33b-單流模型同步生成最長-15-秒立體聲影片-ee992bda</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-10T00:03:16.651Z</news:publication_date>
      <news:title>MiniMax 開放 H3 基礎權重：33B 單流模型同步生成最長 15 秒立體聲影片</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/harnessopt-bench-測試模型改寫代理外殼-模型選擇的影響約為編碼工具-1-8-倍-76ad33a2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T18:01:56.310Z</news:publication_date>
      <news:title>HarnessOpt-Bench 測試模型改寫代理外殼，模型選擇的影響約為編碼工具 1.8 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/mist-測出-23-款模型皆受錯誤提示影響-scope-將-qwen3-4b-答案翻轉率由-35-0-降至-16-3-7d5c379a</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T22:03:54.150Z</news:publication_date>
      <news:title>MIST 測出 23 款模型皆受錯誤提示影響，SCOPE 將 Qwen3-4B 答案翻轉率由 35.0% 降至 16.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/ts-rag-以參考-token-融合歷史序列-六項預測資料集平均-mse-降至-0-310-f32d2c7e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T20:02:35.854Z</news:publication_date>
      <news:title>TS-RAG 以參考 token 融合歷史序列，六項預測資料集平均 MSE 降至 0.310</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/影片模型在高頻多事件區僅答對-0-2-增加取樣幀仍未找回事件證據-e1858030</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T16:02:42.951Z</news:publication_date>
      <news:title>影片模型在高頻多事件區僅答對 0.2%，增加取樣幀仍未找回事件證據</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dash-依推理分歧動態分配蒸餾權重-qwen3-1-7b-數學平均提高-3-2-分-e345f9c0</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T00:02:14.160Z</news:publication_date>
      <news:title>DASH 依推理分歧動態分配蒸餾權重，Qwen3-1.7B 數學平均提高 3.2 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/iarcs-讓-llm-生成-3d-場景獎勵程式-物件碰撞率由-52-67-降至-40-45-8e8724ba</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T00:02:59.008Z</news:publication_date>
      <news:title>iARCS 讓 LLM 生成 3D 場景獎勵程式，物件碰撞率由 52.67% 降至 40.45%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/tutormoments-重播真實教學決策-測出七款-llm-預設傾向過度提示-225a2f55</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-09T02:03:29.225Z</news:publication_date>
      <news:title>TutorMoments 重播真實教學決策，測出七款 LLM 預設傾向過度提示</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/otter-以棋步歷史與時限預測人類決策-15-3m-參數模型報告-55-23-top-1-8e299bdd</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T22:03:49.787Z</news:publication_date>
      <news:title>Otter 以棋步歷史與時限預測人類決策，15.3M 參數模型報告 55.23% top-1</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/smrc-sd-先核對代理狀態再蒸餾成功軌跡-qwen3-1-7b-任務成功率提高約-12-個百分點-cd056bef</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T22:02:40.250Z</news:publication_date>
      <news:title>SMRC-SD 先核對代理狀態再蒸餾成功軌跡，Qwen3-1.7B 任務成功率提高約 12 個百分點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/era-回溯疫情預測的-11-優勢遭重估-資料修訂洩漏可解釋幾乎全部增益-eedf9a1c</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T18:05:27.289Z</news:publication_date>
      <news:title>ERA 回溯疫情預測的 11% 優勢遭重估，資料修訂洩漏可解釋幾乎全部增益</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cloudflare-推出-kitesurf-以輕量-web-引擎取代-chromium-代理瀏覽資源用量降至約三分之一以下-edd4cc38</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T18:04:21.270Z</news:publication_date>
      <news:title>Cloudflare 推出 Kitesurf：以輕量 Web 引擎取代 Chromium，代理瀏覽資源用量降至約三分之一以下</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/herald-揭露搜尋代理可引用未檢索段落-單一成員檢查將實測攻擊率降至零-43b4aa38</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T16:04:28.071Z</news:publication_date>
      <news:title>HERALD 揭露搜尋代理可引用未檢索段落，單一成員檢查將實測攻擊率降至零</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/liquid-ai-開放-lfm2-5-2-6b-權重-2-69b-端側模型支援-128k-上下文與工具呼叫-f7a38588</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T16:03:21.742Z</news:publication_date>
      <news:title>Liquid AI 開放 LFM2.5-2.6B 權重，2.69B 端側模型支援 128K 上下文與工具呼叫</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/cipo-以遮蔽檢索證據分配步驟獎勵-qwen2-5-7b-七項問答平均-f1-升至-0-504-95d8c521</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T14:03:39.660Z</news:publication_date>
      <news:title>CIPO 以遮蔽檢索證據分配步驟獎勵，Qwen2.5-7B 七項問答平均 F1 升至 0.504</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/合成臨床資料通過效用門檻仍可有-79-44-缺失-兩項修訂卻拉大來源分布差距-0eeee11d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T14:04:38.417Z</news:publication_date>
      <news:title>合成臨床資料通過效用門檻仍可有 79.44% 缺失，兩項修訂卻拉大來源分布差距</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/u-opsd-只靠模型內部投票訓練-qwen3-8b-五項數學平均提高-10-7-分-45319a10</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T12:03:13.505Z</news:publication_date>
      <news:title>u-OPSD 只靠模型內部投票訓練，Qwen3-8B 五項數學平均提高 10.7 分</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/bakron-將雙側-hessian-量化降至三次複雜度-8192-方陣核心較-yaqa-快-60-倍-5ea25158</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T12:04:02.715Z</news:publication_date>
      <news:title>BaKron 將雙側 Hessian 量化降至三次複雜度，8192 方陣核心較 YAQA 快 60 倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/690-項代理技能實測-混合檢索-hit-5-達-73-5-加入-llm-知識圖譜反降-11-2-點-897db420</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T08:04:05.835Z</news:publication_date>
      <news:title>690 項代理技能實測：混合檢索 hit@5 達 73.5%，加入 LLM 知識圖譜反降 11.2 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/hyper-es-先用梯度建立低維搜尋空間-數學推理平均高於-grpo-lora-1-ec6954f5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T10:03:26.435Z</news:publication_date>
      <news:title>Hyper-ES 先用梯度建立低維搜尋空間，數學推理平均高於 GRPO-LoRA 1%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/finevo-bench-以-120-項金融任務測代理跨任務演化-codex-配對增益-19-37-分-3a9453cc</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-08T08:03:05.532Z</news:publication_date>
      <news:title>FinEvo-Bench 以 120 項金融任務測代理跨任務演化，Codex 配對增益 19.37 分</news:title>
    </news:news>
  </url>
</urlset>
