READ-ONLY PACKAGE PREVIEW

pydantic-models-py/source-context/tests/scenarios/pydantic-models-py/vally/eval.yaml

Version 354361d83247.bb1 · MIT. This preview displays packaged text and does not execute code. Treat the contents as untrusted instructions.

← Return to resource and package checksum

name: pydantic-models-py-eval
description: Vally evaluation suite for pydantic-models-py
version: "1.0"
type: capability
scoring:
  weights: {}
  threshold: 0.75
defaults:
  runs: 1
  timeout: 30m
  model: claude-sonnet-4.6
  judge_model: gpt-5.5
stimuli:
  - name: base_create_response_indb
    prompt: |-
      Create a complete set of Pydantic models for a "Document" resource following the 
      multi-model pattern. Include Base, Create, Update, Response, and InDB models.
      Use Field constraints, proper inheritance, and camelCase aliases with populate_by_name.
      Save the complete model code to `models/document.py` in the current workspace.

    rubric:
      - Defines Base, Create, Update, Response, and InDB Pydantic models for Document.
      - Uses Field constraints and inheritance appropriately for the multi-model pattern.
      - Implements camelCase aliases and enables populate_by_name.
      - Provides complete executable Python model definitions that match the requested hierarchy.
      - Uses modern Python conventions and readable structure.
    graders:
      - type: run-command
        name: Python syntax check
        weight: 1
        config:
          command: node
          args:
            - .vally/tools/check-python-syntax.mjs
            - .vally/syntax-check-config.json
          timeout: 90s
      - type: run-command
        name: Idiomatic Python quality
        weight: 1
        config:
          command: node
          args:
            - .vally/tools/check-python-idiomatic.mjs
            - .vally/syntax-check-config.json
          timeout: 90s
      - type: prompt
        name: Rubric LLM judge
        weight: 1
        config:
          prompt: |-
            Evaluate the generated answer against the stimulus rubric.
            Use the rubric criteria as the primary judgment basis.
            Score 1 if the answer is missing required code output or is largely non-compliant.
            Score 5 only when the response fully satisfies rubric expectations with correct, executable, idiomatic code.
          scoring: scale_1_5

    constraints:
      expect_skills:
        - pydantic-models-py
  - name: camel_case_aliases
    prompt: |-
      Create a Pydantic model for a "Project" resource with snake_case fields but camelCase 
      aliases for API responses. Include created_at and updated_at with aliases.
      Ensure populate_by_name is True to accept both formats.
      Save the Python code to `models/project.py` in the current workspace.

    rubric:
      - Defines a Project Pydantic model with snake_case field names.
      - Implements camelCase aliases for API responses, including createdAt and updatedAt.
      - Enables populate_by_name so both snake_case and camelCase inputs are accepted.
      - Provides complete executable Python model definitions for the requested fields and aliases.
      - Uses modern Python conventions and readable structure.
    graders:
      - type: run-command
        name: Python syntax check
        weight: 1
        config:
          command: node
          args:
            - .vally/tools/check-python-syntax.mjs
            - .vally/syntax-check-config.json
          timeout: 90s
      - type: run-command
        name: Idiomatic Python quality
        weight: 1
        config:
          command: node
          args:
            - .vally/tools/check-python-idiomatic.mjs
            - .vally/syntax-check-config.json
          timeout: 90s
      - type: prompt
        name: Rubric LLM judge
        weight: 1
        config:
          prompt: |-
            Evaluate the generated answer against the stimulus rubric.
            Use the rubric criteria as the primary judgment basis.
            Score 1 if the answer is missing required code output or is largely non-compliant.
            Score 5 only when the response fully satisfies rubric expectations with correct, executable, idiomatic code.
          scoring: scale_1_5

    constraints:
      expect_skills:
        - pydantic-models-py
environment:
  skills:
    - ../../../../.github/plugins/azure-sdk-python/skills/pydantic-models-py
  files:
    - src: ../../_shared/vally/tools/check-python-syntax.mjs
      dest: .vally/tools/check-python-syntax.mjs
    - src: ../../_shared/vally/tools/check-python-idiomatic.mjs
      dest: .vally/tools/check-python-idiomatic.mjs
    - src: ./syntax-check-config.json
      dest: .vally/syntax-check-config.json
artifacts:
  include:
    - "**/*.py"
    - "**/requirements*.txt"
    - "**/pyproject.toml"
    - "**/setup.py"
    - "**/README.md"
  exclude:
    - "**/.venv/**"
    - "**/venv/**"
    - "**/__pycache__/**"
    - "**/*.pyc"
    - "**/.mypy_cache/**"
    - "**/.pytest_cache/**"