Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
72 changes: 71 additions & 1 deletion src/AiClient.php
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
use Psr\SimpleCache\CacheInterface;
use WordPress\AiClient\Builders\EmbeddingBuilder;
use WordPress\AiClient\Builders\PromptBuilder;
use WordPress\AiClient\Builders\TextExtractionBuilder;
use WordPress\AiClient\Common\Exception\InvalidArgumentException;
use WordPress\AiClient\Common\Exception\RuntimeException;
use WordPress\AiClient\Providers\Contracts\ProviderAvailabilityInterface;
Expand All @@ -19,6 +20,7 @@
use WordPress\AiClient\Results\DTO\Embedding;
use WordPress\AiClient\Results\DTO\EmbeddingResult;
use WordPress\AiClient\Results\DTO\GenerativeAiResult;
use WordPress\AiClient\Results\DTO\TextExtractionResult;

/**
* Main AI Client class providing both fluent and traditional APIs for AI operations.
Expand Down Expand Up @@ -84,6 +86,7 @@
* @phpstan-import-type Prompt from PromptBuilder
* @phpstan-import-type EmbeddingInput from EmbeddingBuilder
* @phpstan-import-type ProviderModelTuple from ModelResolver
* @phpstan-import-type DocumentInput from TextExtractionBuilder
*
* phpcs:ignore Generic.Files.LineLength.TooLong
*/
Expand Down Expand Up @@ -271,6 +274,31 @@ public static function input($input = null, ?ProviderRegistry $registry = null):
);
}

/**
* Creates a text extraction builder for fluent document text extraction (OCR / document parsing).
*
* Chain extractTextResult() to get the structured, per-page result, or extractText() to get
* the extracted content as a single markdown string.
*
* @since n.e.x.t
*
* @param DocumentInput|null $document Optional document to extract text from: a File instance,
* or a string URL, data URI, base64 data, or local file path.
* When the MIME type cannot be inferred from the input (e.g.
* an extensionless URL), pass a File instance with an explicit
* MIME type, or use the builder's withDocument($document, $mimeType).
* @param ProviderRegistry|null $registry Optional custom registry. If null, uses default.
* @return TextExtractionBuilder The text extraction builder instance.
*/
public static function document($document = null, ?ProviderRegistry $registry = null): TextExtractionBuilder
{
return new TextExtractionBuilder(
$registry ?? self::defaultRegistry(),
$document,
self::$eventDispatcher
);
}

/**
* Generates content using a unified API that automatically detects model capabilities.
*
Expand Down Expand Up @@ -492,6 +520,48 @@ public static function generateEmbeddings(
->generateEmbeddings();
}

/**
* Extracts text from a document using the traditional API approach.
*
* @since n.e.x.t
*
* @param DocumentInput $document The document to extract text from.
* @param ModelInterface|ModelConfig|null $modelOrConfig Optional specific model to use,
* or model configuration for auto-discovery,
* or null for defaults.
* @param ProviderRegistry|null $registry Optional custom registry. If null, uses default.
* @return TextExtractionResult The structured extraction result.
*/
public static function extractTextResult(
$document,
$modelOrConfig = null,
?ProviderRegistry $registry = null
): TextExtractionResult {
self::validateModelOrConfigParameter($modelOrConfig);
return self::applyModelOrConfig(self::document($document, $registry), $modelOrConfig)
->extractTextResult();
}

/**
* Extracts text from a document and returns it as a single markdown string.
*
* @since n.e.x.t
*
* @param DocumentInput $document The document to extract text from.
* @param ModelInterface|ModelConfig|null $modelOrConfig Optional specific model to use,
* or model configuration for auto-discovery,
* or null for defaults.
* @param ProviderRegistry|null $registry Optional custom registry. If null, uses default.
* @return string The extracted content, pages joined in order.
*/
public static function extractText(
$document,
$modelOrConfig = null,
?ProviderRegistry $registry = null
): string {
return self::extractTextResult($document, $modelOrConfig, $registry)->toMarkdown();
}

/**
* Creates a new message builder for fluent API usage.
*
Expand Down Expand Up @@ -604,7 +674,7 @@ private static function getConfiguredEmbeddingBuilder(
* Works with any builder that exposes the shared model resolution methods
* (see {@see \WordPress\AiClient\Builders\Traits\ModelResolutionTrait}).
*
* @template T of PromptBuilder
* @template T of PromptBuilder|TextExtractionBuilder
*
* @param T $builder The builder to configure.
* @param ModelInterface|ModelConfig|null $modelOrConfig Specific model, model configuration,
Expand Down
210 changes: 210 additions & 0 deletions src/Builders/TextExtractionBuilder.php
Original file line number Diff line number Diff line change
@@ -0,0 +1,210 @@
<?php

declare(strict_types=1);

namespace WordPress\AiClient\Builders;

use Psr\EventDispatcher\EventDispatcherInterface;
use WordPress\AiClient\Builders\Traits\ModelResolutionTrait;
use WordPress\AiClient\Common\Exception\InvalidArgumentException;
use WordPress\AiClient\Common\Exception\RuntimeException;
use WordPress\AiClient\Events\AfterExtractTextEvent;
use WordPress\AiClient\Events\BeforeExtractTextEvent;
use WordPress\AiClient\Files\DTO\File;
use WordPress\AiClient\Providers\ModelResolver;
use WordPress\AiClient\Providers\Models\DTO\ModelConfig;
use WordPress\AiClient\Providers\Models\DTO\ModelRequirements;
use WordPress\AiClient\Providers\Models\Enums\CapabilityEnum;
use WordPress\AiClient\Providers\Models\TextExtraction\Contracts\TextExtractionModelInterface;
use WordPress\AiClient\Providers\ProviderRegistry;
use WordPress\AiClient\Results\DTO\TextExtractionResult;

/**
* Fluent builder for extracting text from documents (OCR / document parsing).
*
* Text extraction transforms a document into structured, per-page content rather than generating
* a conversational response. Model selection and configuration are shared with
* {@see PromptBuilder} via the {@see ModelResolutionTrait}.
*
* @since n.e.x.t
*
* @phpstan-type DocumentInput string|File
*/
class TextExtractionBuilder
{
use ModelResolutionTrait;

/**
* @var File|null The document to extract text from.
*/
protected ?File $document = null;

/**
* @var EventDispatcherInterface|null The event dispatcher for extraction lifecycle events.
*/
private ?EventDispatcherInterface $eventDispatcher;

/**
* Constructor.
*
* @since n.e.x.t
*
* @param ProviderRegistry $registry The provider registry for finding suitable models.
* @param DocumentInput|null $document Optional initial document to extract text from.
* @param EventDispatcherInterface|null $eventDispatcher Optional event dispatcher for lifecycle events.
*/
public function __construct(
ProviderRegistry $registry,
$document = null,
?EventDispatcherInterface $eventDispatcher = null
) {
$this->modelConfig = new ModelConfig();
$this->modelResolver = new ModelResolver($registry);
$this->eventDispatcher = $eventDispatcher;

if ($document !== null) {
$this->withDocument($document);
}
}

/**
* Creates a deep clone of this builder.
*
* Clones the document and model configuration. Service objects (resolver, event dispatcher)
* are intentionally NOT cloned as they are shared dependencies.
*
* @since n.e.x.t
*/
public function __clone()
{
if ($this->document !== null) {
$this->document = clone $this->document;
}

$this->modelConfig = clone $this->modelConfig;
$this->modelResolver = clone $this->modelResolver;
}

/**
* Sets the document to extract text from.
*
* @since n.e.x.t
*
* @param DocumentInput $document The document: a File instance, or a string URL,
* data URI, base64 data, or local file path.
* @param string|null $mimeType Optional MIME type of the document. Required when it
* cannot be inferred from the input (e.g. an extensionless
* URL such as `https://arxiv.org/pdf/1805.04770`). Ignored
* when a File instance is given.
* @return self
* @throws InvalidArgumentException If the document input is invalid, or if its MIME type is
* not supported for text extraction.
*/
public function withDocument($document, ?string $mimeType = null): self
{
if (is_string($document)) {
$document = new File($document, $mimeType);
}

if (!$document instanceof File) {
throw new InvalidArgumentException('Document must be a File instance or a string.');
}

// Reject unsupported MIME types here, where the file is still the caller's own input, so
// that the error names the offending type instead of surfacing later as a resolution or
// provider failure.
ModelRequirements::extractionInputModality($document);

$this->document = $document;

return $this;
}

/**
* Checks whether the current document and configuration are supported by an available model.
*
* @since n.e.x.t
*
* @return bool True if a suitable text extraction model is available.
*/
public function isSupported(): bool
{
if ($this->document === null) {
return false;
}

$requirements = ModelRequirements::fromExtractionData($this->document, $this->modelConfig);

return $this->modelResolver->isSupported($requirements);
}

/**
* Extracts text from the configured document.
*
* @since n.e.x.t
*
* @return TextExtractionResult The structured extraction result.
* @throws InvalidArgumentException If no document is configured or model validation fails.
* @throws RuntimeException If the resolved model doesn't support text extraction.
*/
public function extractTextResult(): TextExtractionResult
{
$document = $this->document;

if ($document === null) {
throw new InvalidArgumentException(
'Cannot extract text without a document. Add one using withDocument().'
);
}

$capability = CapabilityEnum::textExtraction();
$requirements = ModelRequirements::fromExtractionData($document, $this->modelConfig);
$model = $this->modelResolver->resolve($requirements, $this->modelConfig);

if (!$model instanceof TextExtractionModelInterface) {
throw new RuntimeException(
sprintf(
'Model "%s" does not support text extraction.',
$model->metadata()->getId()
)
);
}

$this->dispatchEvent(new BeforeExtractTextEvent($document, $model, $capability));

$result = $model->extractTextResult($document);

$this->dispatchEvent(new AfterExtractTextEvent($document, $model, $capability, $result));

return $result;
}

/**
* Extracts text from the configured document and returns it as a single markdown string.
*
* @since n.e.x.t
*
* @return string The extracted content, pages joined in order.
* @throws InvalidArgumentException If no document is configured or model validation fails.
* @throws RuntimeException If the resolved model doesn't support text extraction.
*/
public function extractText(): string
{
return $this->extractTextResult()->toMarkdown();
}

/**
* Dispatches an event if an event dispatcher is registered.
*
* @since n.e.x.t
*
* @param object $event The event to dispatch.
* @return void
*/
private function dispatchEvent(object $event): void
{
if ($this->eventDispatcher !== null) {
$this->eventDispatcher->dispatch($event);
}
}
}
Loading
Loading