{
  "id": "classification-evaluate",
  "name": "Classification evaluation evidence",
  "description": "Evaluate up to 1,000 caller-supplied single-label expected/predicted pairs: confusion matrix, per-label precision/recall/F1/support, macro/weighted/micro summaries, balanced accuracy and mismatch evidence. Undefined per-label values use zero with explicit flags; labels are not verified ground truth.",
  "category": "data-assurance",
  "priceUsd": "0.01",
  "pricingStatus": "proposed-unverified",
  "pricing": {
    "unit": "one successful operation call",
    "proposedNominalUsd": "0.01",
    "sixDecimalTokenBaseUnits": "10000",
    "subscription": false,
    "includesPayerWalletOrNetworkFees": false,
    "liveQuoteVerified": false,
    "condition": "Actual SDK challenge is authoritative only within the caller's explicit authorization; configured six-decimal token peg is an operator assertion, not a conversion guarantee."
  },
  "intents": [
    "calculate confusion matrix precision recall f1 from evaluation labels",
    "score agent routing classification predictions against supplied labels",
    "inspect macro micro weighted metrics and error samples"
  ],
  "whenToUse": [
    "calculate confusion matrix precision recall f1 from evaluation labels",
    "score agent routing classification predictions against supplied labels",
    "inspect macro micro weighted metrics and error samples"
  ],
  "whenNotToUse": [
    "Train or execute a model or infer factual ground truth",
    "Multilabel classification probability calibration or deployment certification"
  ],
  "capabilities": [
    "single-label classification metrics",
    "confusion matrix evidence",
    "undefined metric flags",
    "bounded mismatch samples"
  ],
  "related": [
    "ranking-evaluate",
    "span-evaluate"
  ],
  "limits": {
    "inputBytes": 100000,
    "outputBytes": 400000,
    "jsonNodes": 12000,
    "depth": 20,
    "evidence": 200,
    "keyWorkBytes": 3000000,
    "matchingSteps": 2000000,
    "samples": 1000,
    "labels": 50
  },
  "errorCodes": [
    "INPUT_LIMIT",
    "OUTPUT_LIMIT",
    "COMPLEXITY_LIMIT",
    "UNSAFE_KEY",
    "INVALID_NUMBER",
    "INVALID_POINTER",
    "DUPLICATE_ID",
    "DUPLICATE_FIELD",
    "INVALID_RULE",
    "KEY_WORK_LIMIT",
    "LABEL_LIMIT",
    "UNJUDGED_DOCUMENT",
    "INVALID_SPAN",
    "DUPLICATE_SPAN",
    "INVALID_UNICODE",
    "MATCHING_LIMIT",
    "INVALID_CADENCE",
    "INVALID_INTERVAL",
    "NUMERIC_OVERFLOW"
  ],
  "requirementsProfile": {
    "format": "declared-requirements-v1",
    "facts": {
      "execution.suppliedCode": false,
      "execution.remoteMutation": false,
      "execution.llmInference": false,
      "execution.paidMcp": false,
      "verification.semanticTruth": false,
      "verification.sourceAuthenticity": false,
      "verification.liveVulnerabilities": false,
      "numbers.arbitraryPrecisionJson": false,
      "numbers.model": "ieee754-binary64",
      "privacy.requestBodyPersisted": false,
      "execution.deterministic": true,
      "execution.networkAccess": false,
      "privacy.resultBodyPersisted": false,
      "payment.x402": true,
      "payment.mpp": true,
      "operation.id": "classification-evaluate",
      "operation.category": "data-assurance",
      "limit.httpRequestBytes": 131072,
      "limit.requestBytes": 100000,
      "limit.responseBytes": 524288,
      "limit.resultBytes": 400000,
      "limit.jsonDepth": 20,
      "limit.jsonNodes": 12000,
      "input.samples.maxItems": 1000,
      "input.labels.maxItems": 50
    },
    "unknownPolicy": "Undeclared requirements are unknown, never compatible. Matching declared facts does not establish semantic fit or input validity.",
    "preflight": "/preflight"
  },
  "documentation": "/reference/tools/classification-evaluate.md",
  "serviceContract": "/reference/services/classification-evaluate.json",
  "errors": [
    {
      "status": 400,
      "meaning": "Malformed JSON, missing/invalid idempotency key, or payment identifier mismatch",
      "retry": "Correct the request before payment"
    },
    {
      "status": 402,
      "meaning": "Payment challenge or rejected payment",
      "retry": "Use official protocol SDK; inspect payment outcome before another payment"
    },
    {
      "status": 409,
      "meaning": "Idempotency conflict, duplicate proof, or PAYMENT_UNCERTAIN",
      "retry": "Keep original key, body, and proof; reconcile uncertainty with operator; never blindly repay"
    },
    {
      "status": 413,
      "meaning": "Input or generated output too large",
      "retry": "Reduce input; no payment attempted for validation failure"
    },
    {
      "status": 415,
      "meaning": "Unsupported media type or compression",
      "retry": "Send uncompressed application/json"
    },
    {
      "status": 422,
      "meaning": "Schema or service-specific semantic validation failure",
      "retry": "Correct input using returned error code; no payment attempted"
    },
    {
      "status": 429,
      "meaning": "Request/payment-attempt rate exceeded",
      "retry": "Wait for rate limit window; preserve existing payment identity"
    },
    {
      "status": 503,
      "meaning": "Payment configuration/provider/state unavailable, or live DNS preparation failed before settlement",
      "retry": "Check readiness; DNS preparation failures may retry the identical key/body/credential only; uncertainty requires reconciliation"
    }
  ],
  "numericPrecision": "JavaScript IEEE-754 numbers; use strings for large integer IDs/exact decimals where the schema accepts strings. No lossless numeric parsing.",
  "paymentWorkflow": {
    "discoveryOnly": false,
    "supportedProtocols": [
      "x402",
      "mpp"
    ],
    "x402": {
      "credentialHeader": "PAYMENT-SIGNATURE",
      "challengeHeader": "PAYMENT-REQUIRED",
      "receiptHeader": "PAYMENT-RESPONSE",
      "version": 2,
      "scheme": "exact",
      "paymentIdentifier": "payment-identifier extension MUST equal the HTTP Idempotency-Key",
      "sdk": "@x402/core with @x402/evm"
    },
    "mpp": {
      "supported": true,
      "credentialHeader": "Authorization",
      "challengeHeader": "WWW-Authenticate",
      "receiptHeader": "Payment-Receipt",
      "method": "tempo",
      "intent": "charge",
      "sdk": "mppx",
      "tokenDecimals": 6
    },
    "steps": [
      "Check configured readiness and the exact service schema",
      "Generate a fresh random Idempotency-Key for this operation; never use a discovery probe fixture for purchases",
      "Send valid input without a credential to obtain the official protocol challenge",
      "Use the official SDK and authorized wallet to fulfill the challenge",
      "Retry only with identical key, body, protocol and credential",
      "On PAYMENT_UNCERTAIN stop and request operator reconciliation; never blindly pay again"
    ],
    "versionedRetries": "Request fingerprint includes service release version. Retries across a version upgrade can conflict; coordinate upgrades outside the 24-hour replay window and reconcile pending attempts.",
    "docs": "/llms.txt"
  },
  "method": "POST",
  "paths": {
    "x402": "/v1/x402/classification-evaluate",
    "mpp": "/v1/mpp/classification-evaluate"
  },
  "inputSchema": {
    "$schema": "https://json-schema.org/draft/2020-12/schema",
    "type": "object",
    "properties": {
      "samples": {
        "maxItems": 1000,
        "type": "array",
        "items": {
          "type": "object",
          "properties": {
            "id": {
              "type": "string",
              "minLength": 1,
              "maxLength": 80
            },
            "expected": {
              "type": "string",
              "minLength": 1,
              "maxLength": 80
            },
            "predicted": {
              "type": "string",
              "minLength": 1,
              "maxLength": 80
            }
          },
          "required": [
            "id",
            "expected",
            "predicted"
          ],
          "additionalProperties": false
        }
      },
      "labels": {
        "default": [],
        "maxItems": 50,
        "type": "array",
        "items": {
          "type": "string",
          "minLength": 1,
          "maxLength": 80
        }
      }
    },
    "required": [
      "samples"
    ],
    "additionalProperties": false
  },
  "outputSchema": {
    "type": "object",
    "required": [
      "operation",
      "version",
      "result",
      "provenance"
    ],
    "properties": {
      "operation": {
        "const": "classification-evaluate",
        "type": "string"
      },
      "version": {
        "const": "0.29.0",
        "type": "string"
      },
      "result": {
        "$schema": "https://json-schema.org/draft/2020-12/schema",
        "type": "object",
        "properties": {
          "profile": {
            "type": "string",
            "const": "single-label-evaluation-v1"
          },
          "sampleCount": {
            "type": "integer",
            "minimum": 0,
            "maximum": 9007199254740991
          },
          "labels": {
            "type": "array",
            "items": {
              "type": "string"
            }
          },
          "confusionMatrix": {
            "type": "array",
            "items": {
              "type": "array",
              "items": {
                "type": "integer",
                "minimum": 0,
                "maximum": 9007199254740991
              }
            }
          },
          "perLabel": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "precision": {
                  "type": "number"
                },
                "recall": {
                  "type": "number"
                },
                "f1": {
                  "type": "number"
                },
                "label": {
                  "type": "string"
                },
                "truePositive": {
                  "type": "integer",
                  "minimum": 0,
                  "maximum": 9007199254740991
                },
                "falsePositive": {
                  "type": "integer",
                  "minimum": 0,
                  "maximum": 9007199254740991
                },
                "falseNegative": {
                  "type": "integer",
                  "minimum": 0,
                  "maximum": 9007199254740991
                },
                "support": {
                  "type": "integer",
                  "minimum": 0,
                  "maximum": 9007199254740991
                },
                "predictedCount": {
                  "type": "integer",
                  "minimum": 0,
                  "maximum": 9007199254740991
                },
                "precisionDefined": {
                  "type": "boolean"
                },
                "recallDefined": {
                  "type": "boolean"
                },
                "f1Defined": {
                  "type": "boolean"
                }
              },
              "required": [
                "precision",
                "recall",
                "f1",
                "label",
                "truePositive",
                "falsePositive",
                "falseNegative",
                "support",
                "predictedCount",
                "precisionDefined",
                "recallDefined",
                "f1Defined"
              ],
              "additionalProperties": false
            }
          },
          "accuracy": {
            "type": [
              "number",
              "null"
            ]
          },
          "balancedAccuracy": {
            "type": [
              "number",
              "null"
            ]
          },
          "macro": {
            "anyOf": [
              {
                "type": "object",
                "properties": {
                  "precision": {
                    "type": "number"
                  },
                  "recall": {
                    "type": "number"
                  },
                  "f1": {
                    "type": "number"
                  }
                },
                "required": [
                  "precision",
                  "recall",
                  "f1"
                ],
                "additionalProperties": false
              },
              {
                "type": "null"
              }
            ]
          },
          "weighted": {
            "anyOf": [
              {
                "type": "object",
                "properties": {
                  "precision": {
                    "type": "number"
                  },
                  "recall": {
                    "type": "number"
                  },
                  "f1": {
                    "type": "number"
                  }
                },
                "required": [
                  "precision",
                  "recall",
                  "f1"
                ],
                "additionalProperties": false
              },
              {
                "type": "null"
              }
            ]
          },
          "micro": {
            "anyOf": [
              {
                "type": "object",
                "properties": {
                  "precision": {
                    "type": "number"
                  },
                  "recall": {
                    "type": "number"
                  },
                  "f1": {
                    "type": "number"
                  }
                },
                "required": [
                  "precision",
                  "recall",
                  "f1"
                ],
                "additionalProperties": false
              },
              {
                "type": "null"
              }
            ]
          },
          "mismatchCount": {
            "type": "integer",
            "minimum": 0,
            "maximum": 9007199254740991
          },
          "mismatches": {
            "type": "array",
            "items": {
              "type": "object",
              "properties": {
                "id": {
                  "type": "string"
                },
                "expected": {
                  "type": "string"
                },
                "predicted": {
                  "type": "string"
                }
              },
              "required": [
                "id",
                "expected",
                "predicted"
              ],
              "additionalProperties": false
            }
          },
          "mismatchesTruncated": {
            "type": "boolean"
          },
          "conventions": {
            "type": "array",
            "items": {
              "type": "string"
            }
          }
        },
        "required": [
          "profile",
          "sampleCount",
          "labels",
          "confusionMatrix",
          "perLabel",
          "accuracy",
          "balancedAccuracy",
          "macro",
          "weighted",
          "micro",
          "mismatchCount",
          "mismatches",
          "mismatchesTruncated",
          "conventions"
        ],
        "additionalProperties": false
      },
      "provenance": {
        "type": "object",
        "required": [
          "inputSha256",
          "outputSha256",
          "deterministic",
          "externalRequests"
        ],
        "properties": {
          "inputSha256": {
            "type": "string",
            "pattern": "^[a-f0-9]{64}$"
          },
          "outputSha256": {
            "type": "string",
            "pattern": "^[a-f0-9]{64}$"
          },
          "deterministic": {
            "const": true
          },
          "externalRequests": {
            "const": 0
          }
        }
      }
    },
    "additionalProperties": false
  },
  "exampleInput": {
    "samples": [
      {
        "id": "e1",
        "expected": "answer",
        "predicted": "answer"
      },
      {
        "id": "e2",
        "expected": "abstain",
        "predicted": "answer"
      },
      {
        "id": "e3",
        "expected": "abstain",
        "predicted": "abstain"
      }
    ],
    "labels": [
      "answer",
      "abstain"
    ]
  },
  "exampleResponse": {
    "operation": "classification-evaluate",
    "version": "0.29.0",
    "result": {
      "profile": "single-label-evaluation-v1",
      "sampleCount": 3,
      "labels": [
        "abstain",
        "answer"
      ],
      "confusionMatrix": [
        [
          1,
          1
        ],
        [
          0,
          1
        ]
      ],
      "perLabel": [
        {
          "precision": 1,
          "recall": 0.5,
          "f1": 0.6666666666666666,
          "label": "abstain",
          "truePositive": 1,
          "falsePositive": 0,
          "falseNegative": 1,
          "support": 2,
          "predictedCount": 1,
          "precisionDefined": true,
          "recallDefined": true,
          "f1Defined": true
        },
        {
          "precision": 0.5,
          "recall": 1,
          "f1": 0.6666666666666666,
          "label": "answer",
          "truePositive": 1,
          "falsePositive": 1,
          "falseNegative": 0,
          "support": 1,
          "predictedCount": 2,
          "precisionDefined": true,
          "recallDefined": true,
          "f1Defined": true
        }
      ],
      "accuracy": 0.6666666666666666,
      "balancedAccuracy": 0.75,
      "macro": {
        "precision": 0.75,
        "recall": 0.75,
        "f1": 0.6666666666666666
      },
      "weighted": {
        "precision": 0.8333333333333334,
        "recall": 0.6666666666666666,
        "f1": 0.6666666666666666
      },
      "micro": {
        "precision": 0.6666666666666666,
        "recall": 0.6666666666666666,
        "f1": 0.6666666666666666
      },
      "mismatchCount": 1,
      "mismatches": [
        {
          "id": "e2",
          "expected": "abstain",
          "predicted": "answer"
        }
      ],
      "mismatchesTruncated": false,
      "conventions": [
        "Confusion matrix rows are expected labels and columns are predicted labels; labels use code-unit lexical order.",
        "Undefined per-label precision, recall or F1 is 0 with explicit defined flags. Macro includes all declared and observed labels; weighted uses expected support; balanced accuracy includes only supported labels.",
        "Supplied expectations are not verified. No training, probability calibration, multilabel matching or inference about deployment quality."
      ]
    },
    "provenance": {
      "inputSha256": "94e0ebdf6418835b5e24379c85e0dcf742165925f5cc787097b124279db65c32",
      "outputSha256": "96304c42a7fa68f16782023dec0075af56722a1a75d391c578fcedb286492dc3",
      "deterministic": true,
      "externalRequests": 0
    }
  },
  "requiredHeaders": {
    "Content-Type": "application/json",
    "Idempotency-Key": "random 16–128 character operation identifier"
  },
  "fixedExample": "/reference/examples/classification-evaluate.json",
  "execution": {
    "deterministic": true,
    "externalRequests": 0,
    "maxExternalRequests": 0,
    "resultSnapshotPersisted": false,
    "fixedExampleIsIllustrativeSnapshot": false,
    "requiresPayment": true,
    "supportsMcpExecution": false
  },
  "releaseSnapshot": {
    "format": "static-release-reference-v1",
    "sourceVersion": "0.29.0",
    "sourceRegistrySha256": "a2b5ec09bf6c38b9d2879d746a4fded374f5928b445377b0270ef6aa8e6cac65",
    "generatedAt": "2026-10-06T15:38:00Z",
    "releaseAcceptance": "not-verified-by-generator",
    "runtimeReadiness": "not-evaluated",
    "livePaymentsVerified": false,
    "indexingVerified": false,
    "apiOrigin": null,
    "notice": "Build-time release reference. Runtime readiness, deployment, payment settlement and external indexing are not evaluated here. Prices are proposed; this file cannot authorize execution or payment."
  }
}
