AI एजेंट · पाठ

अभिकथन-आधारित एजेंट परीक्षण

टूल कॉल, मध्यवर्ती चरणों और अंतिम आउटपुट की संरचना की जाँच करना।

पाठ 3, कुल 4 में से13 चरण

अभिकथन-आधारित एजेंट परीक्षण, CoddyKit पर AI एजेंट का एक निःशुल्क पाठ है। यह 4 में से 3वाँ पाठ है। आप नीचे पूरा पाठ निःशुल्क पढ़ सकते हैं—फिर अंतर्निहित कोड संपादक और 24/7 एआई ट्यूटर के साथ ब्राउज़र में इसका व्यावहारिक अभ्यास कर सकते हैं। यह AI एजेंट सीखने के मार्ग का हिस्सा है और आपकी प्रगति वेब तथा CoddyKit ऐप पर सिंक होती रहती है। AI एजेंट पाठ्यक्रम में कुल 4 पाठ शामिल हैं।

सटीक स्ट्रिंग मिलान से आगे बढ़ना

चूँकि LLM आउटपुट अनियत होते हैं, इसलिए assert response == 'exact text' से उनका परीक्षण करना अस्थिर होता है। इसके बजाय ऐसी जाँच लिखें जो सटीक शब्दों पर निर्भर हुए बिना प्रतिक्रिया की संरचना और आशय की जाँच करे।

उपकरण-कॉल किए जाने की जाँच

फ़ंक्शन-कॉल करने वाले एजेंटों के लिए सबसे विश्वसनीय जाँच यह पुष्टि करना है कि एजेंट ने सही उपकरण को कॉल करने का विकल्प चुना। यह संरचनात्मक जाँच है — यह LLM की विचार-प्रक्रिया के सटीक शब्दों पर निर्भर नहीं करती।

import json
from unittest.mock import patch, MagicMock

@patch('myagent.client.chat.completions.create')
def test_agent_calls_search_tool(mock_create):
    # Mock: agent decides to call search_web
    tool_call = MagicMock()
    tool_call.function.name = 'search_web'
    tool_call.function.arguments = json.dumps({'query': 'Python tutorials'})
    mock_create.return_value = MagicMock(
        choices=[MagicMock(message=MagicMock(tool_calls=[tool_call]))]
    )

    response = mock_create()  # simulating the agent call
    tc = response.choices[0].message.tool_calls

    assert tc is not None
    assert len(tc) > 0
    assert tc[0].function.name == 'search_web'

# --- demo: give unittest.mock.patch a real dotted path to patch ---
import sys
import types

_myagent = types.ModuleType('myagent')
_myagent.client = types.SimpleNamespace(
    chat=types.SimpleNamespace(completions=types.SimpleNamespace(create=lambda *a, **k: None))
)
sys.modules['myagent'] = _myagent

test_agent_calls_search_tool()
print('test_agent_calls_search_tool: PASS')

सही उपकरण नाम की जाँच

केवल उपकरण-कॉल मौजूद होने की जाँच करने के अलावा यह भी सुनिश्चित करें कि विशिष्ट उपकरण नाम अपेक्षित नाम से मेल खाता है। इससे वे स्थितियाँ पकड़ी जाती हैं जिनमें एजेंट किसी प्रश्न के लिए गलत उपकरण चुनता है।

import json
from unittest.mock import MagicMock

def extract_tool_calls(response) -> list:
    message = response.choices[0].message
    if not message.tool_calls:
        return []
    return [
        {
            'name': tc.function.name,
            'args': json.loads(tc.function.arguments)
        }
        for tc in message.tool_calls
    ]

# In a test:
# calls = extract_tool_calls(mock_response)
# assert calls[0]['name'] == 'get_weather'
# assert calls[0]['args']['city'] == 'Paris'
print('Tool name and argument assertions are the most reliable agent tests')

उपकरण तर्कों की जाँच

उपकरण नाम की पुष्टि करने के बाद जाँचें कि उसके तर्क सही हैं। एजेंट को केवल सही उपकरण चुनना ही नहीं, बल्कि उपयोगकर्ता के अनुरोध से सही पैरामीटर लेकर उसे भरना भी आवश्यक है।

import json
from unittest.mock import patch, MagicMock

@patch('myagent.client.chat.completions.create')
def test_weather_tool_gets_correct_city(mock_create):
    tool_call = MagicMock()
    tool_call.function.name = 'get_weather'
    tool_call.function.arguments = json.dumps({'city': 'Tokyo', 'unit': 'celsius'})
    mock_create.return_value = MagicMock(
        choices=[MagicMock(message=MagicMock(tool_calls=[tool_call]))]
    )

    response = mock_create()
    args = json.loads(response.choices[0].message.tool_calls[0].function.arguments)

    assert args['city'] == 'Tokyo'
    assert args.get('unit') in ['celsius', 'fahrenheit', None]  # flexible

# --- demo: give unittest.mock.patch a real dotted path to patch ---
import sys
import types

_myagent = types.ModuleType('myagent')
_myagent.client = types.SimpleNamespace(
    chat=types.SimpleNamespace(completions=types.SimpleNamespace(create=lambda *a, **k: None))
)
sys.modules['myagent'] = _myagent

test_weather_tool_gets_correct_city()
print('test_weather_tool_gets_correct_city: PASS')

आउटपुट के लिए JSON Schema सत्यापन

जब आपका एजेंट संरचित JSON लौटाता है, तो आउटपुट को JSON Schema के आधार पर सत्यापित करें, ताकि यह सुनिश्चित हो सके कि सभी आवश्यक फ़ील्ड मौजूद हैं और उनके प्रकार सही हैं। jsonschema लाइब्रेरी यह काम आसान बनाती है।

# pip install jsonschema
import jsonschema

AGENT_RESPONSE_SCHEMA = {
    'type': 'object',
    'required': ['answer', 'sources', 'confidence'],
    'properties': {
        'answer': {'type': 'string', 'minLength': 1},
        'sources': {
            'type': 'array',
            'items': {'type': 'string', 'format': 'uri'}
        },
        'confidence': {'type': 'number', 'minimum': 0, 'maximum': 1}
    }
}

def test_agent_output_schema(agent_output: dict):
    try:
        jsonschema.validate(instance=agent_output, schema=AGENT_RESPONSE_SCHEMA)
        print('Schema validation passed')
    except jsonschema.ValidationError as e:
        raise AssertionError(f'Invalid agent output: {e.message}')

कीवर्ड की मौजूदगी की जाँच

ऐसी पाठ प्रतिक्रियाओं में जहाँ सटीक शब्द अलग-अलग हो सकते हैं, जाँचें कि मुख्य अवधारणाएँ या शब्द आउटपुट में मौजूद हैं। यह लचीला होने के साथ अर्थपूर्ण भी है — एजेंट के उत्तर में कम-से-कम संबंधित शब्दों का उल्लेख होना चाहिए।

def assert_keywords_present(text: str, keywords: list, require_all: bool = True):
    lower_text = text.lower()
    found = [kw.lower() in lower_text for kw in keywords]

    if require_all:
        missing = [kw for kw, f in zip(keywords, found) if not f]
        assert not missing, f'Missing keywords: {missing}'
    else:
        assert any(found), f'None of {keywords} found in: {text[:100]}'

# Tests
response = 'The capital city of France is Paris, located in western Europe.'
assert_keywords_present(response, ['paris', 'france', 'capital'])
print('All keywords present!')  # passes

assert_keywords_present(response, ['spain', 'france'], require_all=False)
print('At least one keyword present!')  # passes

प्रतिक्रिया प्रारूप की जाँच: प्रकार की जाँच

प्रकार की जाँच तेज़ और विश्वसनीय होती है। सुनिश्चित करें कि एजेंट एक dict लौटाता है, None नहीं, सूची वाले फ़ील्ड सूचियाँ हैं और संख्यात्मक फ़ील्ड मान्य सीमाओं में हैं।

def test_agent_returns_valid_structure(agent_result):
    # Type checks
    assert isinstance(agent_result, dict), 'Result must be a dict'
    assert isinstance(agent_result.get('answer'), str), 'answer must be a string'
    assert isinstance(agent_result.get('steps'), list), 'steps must be a list'

    # Non-empty checks
    assert len(agent_result['answer']) > 0, 'answer must not be empty'
    assert len(agent_result['steps']) >= 1, 'must have at least one step'

    # Range checks
    confidence = agent_result.get('confidence', 0)
    assert 0.0 <= confidence <= 1.0, 'confidence must be 0-1'

print('Structural assertions are fast and reliable')

finish_reason की जाँच

finish_reason फ़ील्ड बताता है कि मॉडल ने निर्माण क्यों रोका। इसकी जाँच से समस्याओं का पता लगाने में मदद मिलती है: 'stop' का अर्थ है साफ़-सुथरा उत्तर, 'tool_calls' का अर्थ है कि एजेंट किसी उपकरण को कॉल करना चाहता है और 'length' का अर्थ है कि उत्तर बीच में काट दिया गया।

from unittest.mock import MagicMock

def test_agent_stops_cleanly(mock_response):
    finish_reason = mock_response.choices[0].finish_reason
    assert finish_reason in ('stop', 'tool_calls'), \
        f'Unexpected finish_reason: {finish_reason}'

def test_no_truncation(mock_response):
    finish_reason = mock_response.choices[0].finish_reason
    assert finish_reason != 'length', \
        'Response was truncated — increase max_tokens'

# Example mock for a clean stop
mock = MagicMock()
mock.choices = [MagicMock(finish_reason='stop')]
test_agent_stops_cleanly(mock)
print('finish_reason: stop — clean termination')

लूप में चरणों की संख्या की जाँच

लूप में चलने वाले एजेंट को उचित संख्या में चरणों के भीतर पूरा हो जाना चाहिए। जाँच करें कि एजेंट अधिकतम पुनरावृत्ति संख्या के भीतर समाप्त हो जाता है — इससे उन अनंत लूपों का पता चलता है जिन्हें आपका max_iterations सुरक्षा-नियंत्रण रोकने के लिए बनाया गया है।

def test_agent_completes_in_bounded_steps(mock_agent):
    result = mock_agent.run('Search for the weather in Paris')

    # Agent should complete within 5 steps
    assert result['steps_taken'] <= 5, \
        f'Agent took too many steps: {result["steps_taken"]}'

    # Agent should produce a final answer, not exit on timeout
    assert result['status'] == 'completed', \
        f'Agent did not complete: {result["status"]}'

    assert result['answer'] is not None

print('Bounding step count prevents runaway agents from passing tests')

कई इनपुट के लिए परीक्षणों का पैरामीटरीकरण

pytest का @pytest.mark.parametrize आपको एक ही परीक्षण को कई अलग-अलग इनपुट के साथ चलाने देता है। यह जाँचने के लिए आदर्श है कि आपका एजेंट विभिन्न प्रकार के प्रश्नों को सही उपकरणों तक भेजता है।

import pytest
from unittest.mock import patch, MagicMock
import json

@pytest.mark.parametrize('query,expected_tool', [
    ('What is the weather in Tokyo?', 'get_weather'),
    ('Calculate 15% of 200', 'calculator'),
    ('Search for Python books', 'web_search'),
    ('What time is it in Berlin?', 'get_time'),
])
@patch('myagent.client.chat.completions.create')
def test_agent_tool_routing(mock_create, query, expected_tool):
    tool_call = MagicMock()
    tool_call.function.name = expected_tool
    tool_call.function.arguments = json.dumps({'input': query})
    mock_create.return_value = MagicMock(
        choices=[MagicMock(message=MagicMock(tool_calls=[tool_call]))]
    )
    response = mock_create()
    actual = response.choices[0].message.tool_calls[0].function.name
    assert actual == expected_tool

कस्टम जाँच सहायक लिखना

जैसे-जैसे आपका एजेंट परीक्षण-संग्रह बढ़ता है, सामान्य जाँच प्रारूपों को सहायक फ़ंक्शन में अलग करें। इससे परीक्षण छोटे, अधिक पठनीय और एजेंट के प्रतिक्रिया प्रारूप में बदलाव होने पर आसानी से अनुरक्षित किए जा सकते हैं।

import json

def assert_tool_called(response, tool_name: str, required_args: dict = None):
    message = response.choices[0].message
    assert message.tool_calls, 'Expected tool call but got plain text'
    names = [tc.function.name for tc in message.tool_calls]
    assert tool_name in names, f'Expected {tool_name}, got {names}'

    if required_args:
        for tc in message.tool_calls:
            if tc.function.name == tool_name:
                args = json.loads(tc.function.arguments)
                for key, val in required_args.items():
                    assert args.get(key) == val, \
                        f'Arg {key}: expected {val}, got {args.get(key)}'

# Clean test using the helper:
# assert_tool_called(response, 'get_weather', {'city': 'Paris'})

# --- demo ---
from unittest.mock import MagicMock

tool_call = MagicMock()
tool_call.function.name = 'get_weather'
tool_call.function.arguments = json.dumps({'city': 'Paris'})
response = MagicMock(choices=[MagicMock(message=MagicMock(tool_calls=[tool_call]))])

assert_tool_called(response, 'get_weather', {'city': 'Paris'})
print('assert_tool_called passed: agent called get_weather with city=Paris')

ज्ञान-जाँच: जाँच-आधारित एजेंट परीक्षण

एजेंट परीक्षणों के लिए जाँच रणनीतियों की अपनी समझ जाँचें।

पुनरावलोकन: जाँच-आधारित एजेंट परीक्षण

अब आपके पास एजेंट परीक्षणों के लिए जाँच का पूरा उपकरण-संग्रह है:

  • जब एजेंट को किसी उपकरण का उपयोग करना चाहिए, तब जाँचें कि tool_calls खाली नहीं है
  • tc.function.name == 'expected_tool' से सही उपकरण नाम की जाँच करें
  • tc.function.arguments को JSON के रूप में पार्स करके उपकरण तर्कों को सत्यापित करें
  • संरचित आउटपुट सत्यापन के लिए jsonschema.validate() का उपयोग करें
  • लचीली पाठ जाँच के लिए कीवर्ड की मौजूदगी की जाँच करें
  • लूप वाले एजेंटों के लिए finish_reason और चरणों की संख्या की जाँच करें
  • कई इनपुट परिदृश्यों के लिए @pytest.mark.parametrize का उपयोग करें
शुरुआत निःशुल्क

एआई शिक्षक के साथ AI एजेंट सीखें — निःशुल्क

अपने ब्राउज़र में वास्तविक कोड लिखें और चलाएँ, चौबीसों घंटे एआई शिक्षक से तुरंत सहायता पाएँ, और वेब या ऐप पर वहीं से शुरू करें जहाँ आपने छोड़ा था।

पाठ्यक्रम
60
पाठ
239

अक्सर पूछे जाने वाले प्रश्न

क्या “अभिकथन-आधारित एजेंट परीक्षण” पाठ निःशुल्क है?

हाँ—“अभिकथन-आधारित एजेंट परीक्षण” का पूरा पाठ यहाँ वेब पर निःशुल्क पढ़ा जा सकता है। इंटरैक्टिव अभ्यास (अंतर्निहित कोड संपादक और 24/7 एआई ट्यूटर) करने और AI एजेंट पाठ्यक्रम का बाकी हिस्सा अनलॉक करने के लिए CoddyKit PRO लें। AI एजेंट पाठ्यक्रम में कुल 4 पाठ शामिल हैं।

“अभिकथन-आधारित एजेंट परीक्षण” में मैं क्या सीखूँगा?

टूल कॉल, मध्यवर्ती चरणों और अंतिम आउटपुट की संरचना की जाँच करना। आप ब्राउज़र में सीधे चलाए जाने वाले व्यावहारिक कोड के साथ AI एजेंट का अभ्यास करते हैं, और पाठ पूरा करते समय 24/7 एआई ट्यूटर आपके प्रश्नों के उत्तर देता है।

क्या AI एजेंट शुरू करने के लिए मुझे किसी अनुभव की आवश्यकता है?

पहले के अनुभव की आवश्यकता नहीं है। CoddyKit पर AI एजेंट शुरुआती से लेकर उन्नत शिक्षार्थियों तक सभी के लिए व्यवस्थित किया गया है, इसलिए आप यहीं से या शुरुआत से सीखना शुरू कर सकते हैं और अपनी गति से आगे बढ़ सकते हैं। यह 4 में से 3वाँ पाठ है।

“अभिकथन-आधारित एजेंट परीक्षण” पाठ पूरा करने में कितना समय लगता है?

CoddyKit का अधिकांश पाठ लगभग 5–10 मिनट में पूरा हो जाता है। हर पाठ छोटा और संवादात्मक है, इसलिए आप लगातार प्रगति करते हैं और वेब या ऐप पर वहीं से सीखना जारी रख सकते हैं जहाँ आपने छोड़ा था।

क्या मैं इस AI एजेंट पाठ में कोड लिख और चला सकता हूँ?

हाँ। हर AI एजेंट पाठ में एक अंतर्निर्मित कोड संपादक शामिल है, जिससे आप सीधे अपने ब्राउज़र में वास्तविक कोड लिख और चला सकते हैं और तुरंत एआई प्रतिक्रिया पा सकते हैं—स्थानीय सेटअप की आवश्यकता नहीं है।

इस पाठ्यक्रम के सभी पाठ

  1. एजेंट का परीक्षण अलग क्यों होता है
  2. परीक्षणों में LLM कॉल का मॉक बनाना
  3. अभिकथन-आधारित एजेंट परीक्षण
  4. एजेंट पाइपलाइनों के एकीकरण परीक्षण
← AI एजेंट पर वापस जाएँ