For Azure Open AI Reinforcement fine tuning, python grader does not work..

Sarvesh Jadhav 5 Reputation points Microsoft Employee
2025-09-30T14:04:49.2266667+00:00

Here is the code I ran to test out a very simple python grader according to documentation on how to test these graders

# validate the grader
import os
import json
import requests
from dotenv import load_dotenv

testGrader = {
    "name": "test_grader",
    "image_tag": "2025-05-08",
    "type": "python",
    "source": "def grade(sample, item):\n    return 0.9\n"
}

load_dotenv(override=True)

AZURE_API_KEY = os.getenv("AZURE_API_KEY")
AZURE_API_ENDPOINT = os.getenv("AZURE_API_ENDPOINT")
headers = {"Authorization": f"Bearer {AZURE_API_KEY}"}

payload = {"grader": testGrader}
response = requests.post(
    f"{AZURE_API_ENDPOINT}/openai/v1/fine_tuning/alpha/graders/validate",
    json=payload,
    headers=headers
)
print("validate request_id:", response.headers["x-request-id"])
print("validate response:", response.text)

# run the grader with a test reference and sample
payload = {
  "grader": testGrader,
  "item": {
     "reference_answer": "test"
  },
  "model_sample": "test"
}
response = requests.post(
    f"{AZURE_API_ENDPOINT}/openai/v1/fine_tuning/alpha/graders/run",
    json=payload,
    headers=headers
)
print("run request_id:", response.headers["x-request-id"])
print("run response:", response.text)

The results indicate that while the first request to validate the grader is successful, the second request to run the grader returns an internal error. The output is 0 instead of the expected 0.9, and the response contains "other_error: true" without any further details.

Below is the output received:

validate request_id: "some id"
validate response: {
  "grader": {
    "name": "test_grader",
    "image_tag": "2025-05-08",
    "type": "python",
    "source": "def grade(sample, item):\n    return 0.9\n"
  }
}
run request_id: "some id"
run response: {
  "reward": 0.0,
  "metadata": {
    "name": "test_grader",
    "type": "python",
    "errors": {
      "formula_parse_error": false,
      "sample_parse_error": false,
      "sample_parse_error_details": null,
      "truncated_observation_error": false,
      "unresponsive_reward_error": false,
      "invalid_variable_error": false,
      "invalid_variable_error_details": null,
      "other_error": true,
      "python_grader_server_error": false,
      "python_grader_server_error_type": null,
      "python_grader_runtime_error": false,
      "python_grader_runtime_error_details": null,
      "model_grader_server_error": false,
      "model_grader_refusal_error": false,
      "model_grader_refusal_error_details": null,
      "model_grader_parse_error": false,
      "model_grader_parse_error_details": null,
      "model_grader_server_error_details": null,
      "endpoint_grader_client_error": false,
      "endpoint_grader_client_error_details": null,
      "endpoint_grader_server_error": false,
      "endpoint_grader_server_error_details": null,
      "endpoint_grader_response_validation_error": false
    },
    "execution_time": 0.00041031837463378906,
    "scores": {},
    "token_usage": null,
    "sampled_model_name": null
  },
  "sub_rewards": {},
  "model_grader_token_usage_per_model": {}
}

I would really appreciate if someone could help me figure out the problem here.

Foundry Tools
Foundry Tools

Formerly known as Azure AI Services or Azure Cognitive Services is a unified collection of prebuilt AI capabilities within the Microsoft Foundry platform


Your answer

Answers can be marked as 'Accepted' by the question author and 'Recommended' by moderators, which helps users know the answer solved the author's problem.