<?xml version="1.0" encoding="UTF-8"?><urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
          <url>
            <loc>https://1cademy.com/node/ch-post-training-alignment-and-calibration-frontier-foundation-models-capability-evaluation-and-just/V7E0LHxoPfDoed886jj0</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/identify-the-rubric-category-that-the-rbrm-classifier-should-assign-to-this-output-and-explain-what-/0AOHvb5uj4rKigirMSvH</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/match-each-input-processed-by-a-rule-based-reward-model-rbrm-to-its-accurate-description/0CkV06lzaYIHhC2LJFKS</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/raw-pre-trained-base-models-inherently-possess-the-capacity-to-accurately-follow-open-ended-instruct/35B4p6vDW9Y1hJYiL2dN</loc>
            <lastmod>2026-09-11T16:41:49.745Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/degradation-of-model-calibration-from-post-training-alignment/4HZi5ZRFd5Tx9iD9INxt</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/origin-of-gpt-exam-capabilities-in-pre-training/4LeTVFfqd96ZmQ2oEFLG</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/how-does-post-training-alignment-such-as-ppo-reinforcement-learning-affect-the-calibration-of-large-/4kzVpcOBOzU1G4weCvQb</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/within-this-alignment-framework-themselves-are-used-as-tools-to-steer-model-behavior/5Xp14HpoTh7foe1YcfpV</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/when-evaluated-on-multiple-choice-standardized-exam-sections-what-average-score-is-achieved-by-the-b/7JYV2i4ZX389yB5ZcJ2I</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/inputs-and-rubric-classification-mechanism-of-rbrms/9RpufnORE83uX1ZCEZZI</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/match-each-standardized-exam-benchmark-metric-with-its-corresponding-empirical-result-for-gpt/9Vox1Ogy6i1B3ROR8e8H</loc>
            <lastmod>2026-09-11T16:30:37.016Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/which-set-of-failure-modes-in-standard-reinforcement-learning-from-human-feedback-rlhf-is-the-model-/A1z3r82qCPRH4o5DLR1F</loc>
            <lastmod>2026-09-11T16:42:39.956Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/rlhf-effects-on-capability-and-calibration-frontier-foundation-models-capability-evaluation-and-just/CbzXftYJKq1k0cg1g6qd</loc>
            <lastmod>2026-09-11T16:30:55.282Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/how-do-the-predicted-probabilities-logprobs-of-pre-trained-models-such-as-gpt-relate-to-task-perform/CrMw3HpU8gfYASDbKwlG</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/during-which-training-phase-do-rule-based-reward-models-rbrms-supply-additional-reward-signals-to-th/FIszfDM2XPAzVMI6XOru</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/which-post-training-processes-develop-a-models-capacity-to-follow-open-ended-instructions-accurately/GMekf3F99GX7PtQmGHcl</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/safety-alignment-and-rule-based-reward-models-frontier-foundation-models-capability-evaluation-and-j/Il7imQOVEKB6eUpHVOh3</loc>
            <lastmod>2026-09-11T16:31:50.981Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/rule-based-reward-models-rbrms-rely-solely-on-scalar-human-preference-models-to-enforce-safety-polic/Qe6DtFS2hJQDXTpTJT9C</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/how-are-rule-based-reward-models-rbrms-categorized-in-terms-of-classifier-type/TF49WgCPT7Xoi9mJ0MXZ</loc>
            <lastmod>2026-09-11T16:43:12.182Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/model-assisted-safety-pipeline/X3py8tIEQkWKmOb0mOTB</loc>
            <lastmod>2026-09-11T16:31:15.684Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/rule-based-reward-models-rbrms/bEi30Hk2cWXzVK6CGkf4</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/within-the-model-assisted-safety-pipeline-what-specific-role-do-rule-based-reward-models-rbrms-perfo/iFiSWSELSnkB3UriqW21</loc>
            <lastmod>2026-09-11T16:42:56.181Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/evaluation-asymmetry-in-base-and-rlhf-free-response-comparison/jNkgmiBAfAAMpji171GP</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/order-the-steps-involved-in-evaluating-a-policy-response-using-a-rule-based-reward-model-rbrm/n6rrzaTzPgDzywl6VUVg</loc>
            <lastmod>2026-09-11T16:44:37.247Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/when-attempting-to-evaluate-pre-trained-base-models-against-post-rlhf-models-on-an-equal-footing-whi/tWJWinGz05qNWMFtKort</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/based-on-empirical-evaluations-on-standardized-exam-benchmarks-what-effect-does-reinforcement-learni/zQMtiHbL1sxJrraksNSb</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url>
          <url>
            <loc>https://1cademy.com/node/explain-the-two-key-safety-objectives-that-rule-based-reward-models-rbrms-are-designed-to-balance-re/zqppaQ4l4hgYNHCM77cr</loc>
            <lastmod>2026-09-11T16:16:06.203Z</lastmod>
            <changefreq>hourly</changefreq>
          </url></urlset>