Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions extract-core/extract_core/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
MarkdownDoc,
OutputFormat,
Pages,
PipelineSize,
Ranges,
Result,
Status,
Expand Down Expand Up @@ -77,6 +78,7 @@
"Ranges",
"Pages",
"Pipeline",
"PipelineSize",
"PipelineType",
"Result",
"Status",
Expand Down
9 changes: 9 additions & 0 deletions extract-core/extract_core/__main__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
from .cli import cli_app


def main() -> None:
cli_app()


if __name__ == "__main__":
main()
46 changes: 46 additions & 0 deletions extract-core/extract_core/cli/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
import importlib
import os
from typing import Annotated

import typer
from icij_common.logging_utils import setup_loggers

import extract_core

from .configs import configs_app
from .utils import AsyncTyper

cli_app = AsyncTyper(
context_settings={"help_option_names": ["-h", "--help"]},
pretty_exceptions_enable=False,
)
cli_app.add_typer(configs_app)


def version_callback(value: bool) -> None: # noqa: FBT001
if value:
package_version = importlib.metadata.version(extract_core.__name__)
print(package_version)
raise typer.Exit()


def pretty_exc_callback(value: bool) -> None: # noqa: FBT001
if not value:
os.environ["TYPER_STANDARD_TRACEBACK"] = "1"


@cli_app.callback()
def main(
version: Annotated[ # noqa: ARG001
bool | None,
typer.Option("--version", callback=version_callback, is_eager=True),
] = None,
*,
pretty_exceptions: Annotated[ # noqa: ARG001
bool,
typer.Option(
"--pretty-exceptions", callback=pretty_exc_callback, is_eager=True
),
] = False,
) -> None:
setup_loggers(["__main__", extract_core.__name__])
37 changes: 37 additions & 0 deletions extract-core/extract_core/cli/configs.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
from typing import Annotated

import typer

from extract_core import PipelineSize

from ..configs import PipelineType
from ..default_config import default_config
from ..objects import Device
from .utils import AsyncTyper

_CONFIGS = "config"

_DEFAULT_CONFIG_HELP = (
"display default configuration for a given pipeline type,"
" device and computing power"
)
_DEFAULT_CONFIG_DEVICE_HELP = "device for which the default config should be displayed"
_DEFAULT_CONFIG_SIZE_HELP = "pipeline size, smaller is faster"

configs_app = AsyncTyper(name=_CONFIGS)


@configs_app.async_command(name="get-default", help=_DEFAULT_CONFIG_HELP)
async def display_default_config(
pipeline_type: PipelineType,
device: Annotated[
Device,
typer.Option("-d", "--device", help=_DEFAULT_CONFIG_DEVICE_HELP),
] = Device.CUDA,
size: Annotated[
PipelineSize,
typer.Option("-s", "--size", help=_DEFAULT_CONFIG_SIZE_HELP),
] = PipelineSize.SMALL,
) -> None:
config = default_config(pipeline_type=pipeline_type, device=device, size=size)
print(config.model_dump_json(indent=2))
25 changes: 25 additions & 0 deletions extract-core/extract_core/cli/utils.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
import asyncio
import sys
from collections.abc import Callable
from functools import wraps
from typing import Any

import typer


class AsyncTyper(typer.Typer):
def async_command(self, *args, **kwargs) -> Callable[[Callable], Callable]:
def decorator(async_func: Callable) -> Callable:
@wraps(async_func)
def sync_func(*_args, **_kwargs) -> Any:
res = asyncio.run(async_func(*_args, **_kwargs))
return res

self.command(*args, **kwargs)(sync_func)
return async_func

return decorator


def eprint(*args, **kwargs) -> None:
print(*args, file=sys.stderr, **kwargs)
48 changes: 48 additions & 0 deletions extract-core/extract_core/default_config.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
from .configs import BasePipelineConfig, Device, PipelineType
from .objects import PipelineSize


def default_config(
*, pipeline_type: PipelineType, device: Device, size: PipelineSize
) -> BasePipelineConfig:
match pipeline_type:
case PipelineType.DOCLING:
return _default_docling_config(device=device, size=size)
case PipelineType.MARKER:
return _default_marker_config(size)
case PipelineType.MINER_U:
return _default_mineru_config(size)
case _:
raise NotImplementedError(f"unsupported pipeline type {pipeline_type}")


def _default_docling_config(
*, device: Device, size: PipelineSize
) -> BasePipelineConfig:
from .docling_ import DoclingPipelineConfig, default_format_opts # noqa: PLC0415

format_opts = default_format_opts(device=device, size=size)
pipeline_cfg = DoclingPipelineConfig(format_options=format_opts)
return pipeline_cfg


def _default_marker_config(size: PipelineSize) -> BasePipelineConfig:
from .marker_ import MarkerPipelineConfig # noqa: PLC0415

mode = "balanced" if size is PipelineSize.LARGE else "fast"
return MarkerPipelineConfig(config={"mode": mode})


def _default_mineru_config(size: PipelineSize) -> BasePipelineConfig:
from .miner_u import MinerUConfig, MinerUPipelineConfig # noqa: PLC0415

match size:
case PipelineSize.SMALL:
tier = "basic"
case PipelineSize.MEDIUM:
tier = "standard"
case PipelineSize.LARGE:
tier = "advanced"
case _:
raise TypeError(f"unsupported pipeline size {size}")
return MinerUPipelineConfig(config=MinerUConfig(tier=tier))
65 changes: 55 additions & 10 deletions extract-core/extract_core/docling_.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,9 @@
import importlib
from copy import deepcopy
from functools import cache
from typing import TYPE_CHECKING, Annotated, Any, ClassVar, get_type_hints

from docling.backend.image_backend import ImageDocumentBackend
from docling.datamodel.backend_options import BackendOptions, BaseBackendOptions
from docling.datamodel.base_models import (
BaseFormatOption,
Expand All @@ -20,6 +22,15 @@
)
from docling.datamodel.settings import DebugSettings
from docling.datamodel.settings import InferenceSettings as DoclingInferenceSettings
from docling.document_converter import (
BoxNoteFormatOption,
CsvFormatOption,
ExcelFormatOption,
OdsFormatOption,
OdtFormatOption,
PowerpointFormatOption,
WordFormatOption,
)
from icij_common.pydantic_utils import (
merge_configs,
safe_copy,
Expand All @@ -31,7 +42,7 @@
from pydantic_core.core_schema import SerializerFunctionWrapHandler

from .configs import BasePipelineConfig, PipelineType, ResultBufferConfig
from .objects import BaseModel, Device, SupportedExt
from .objects import BaseModel, Device, PipelineSize, SupportedExt
from .utils import all_subclasses

if TYPE_CHECKING:
Expand Down Expand Up @@ -233,7 +244,48 @@ def to_docling(self, device: Device) -> BaseFormatOption: # noqa: ANN201
)


def _default_format_opts() -> dict[InputFormat, DoclingFormatOption]:
def default_format_opts(
*, size: PipelineSize = PipelineSize.SMALL, device: Device = Device.CPU
) -> dict[InputFormat, DoclingFormatOption]:
from docling.backend.docling_parse_backend import ( # noqa: PLC0415
ThreadedDoclingParseDocumentBackend,
)
from docling.pipeline.threaded_standard_pdf_pipeline import ( # noqa: PLC0415
ThreadedStandardPdfPipeline,
)
from docling.pipeline.vlm_pipeline import VlmPipeline # noqa: PLC0415

default = _default_format_options()
accelerator_opts = deepcopy(default[InputFormat.PDF].pipeline_options)[
"accelerator_options"
]
accelerator_opts["device"] = device.to_docling()
pdf_pipeline_opts = default[InputFormat.PDF].pipeline_options
pdf_pipeline_opts["accelerator_options"] = accelerator_opts
match size:
case PipelineSize.LARGE:
pipeline = VlmPipeline.__name__
case _:
pipeline = ThreadedStandardPdfPipeline.__name__
backend_opts = default[InputFormat.PDF].backend_options
pdf_fmt_opts = DoclingFormatOption(
pipeline_options=pdf_pipeline_opts,
pipeline_cls=pipeline,
backend=ThreadedDoclingParseDocumentBackend.__name__,
backend_options=backend_opts,
)
default[InputFormat.PDF] = pdf_fmt_opts
image_fmt_opts = DoclingFormatOption(
pipeline_options=deepcopy(pdf_pipeline_opts),
pipeline_cls=pipeline,
backend=ImageDocumentBackend.__name__,
backend_options=deepcopy(backend_opts),
)
default[InputFormat.IMAGE] = image_fmt_opts
return default


def _default_format_options() -> dict[InputFormat, DoclingFormatOption]:
from docling.backend.json.docling_json_backend import ( # noqa: PLC0415
DoclingJSONBackend,
)
Expand All @@ -242,27 +294,20 @@ def _default_format_opts() -> dict[InputFormat, DoclingFormatOption]:
from docling.document_converter import ( # noqa: PLC0415 # noqa: PLC0415
AsciiDocFormatOption,
AudioFormatOption,
BoxNoteFormatOption,
CsvFormatOption,
DclxFormatOption,
EbcdicFormatOption,
EmailFormatOption,
EpubFormatOption,
ExcelFormatOption,
FormatOption,
HTMLFormatOption,
ImageFormatOption,
IWorkPagesFormatOption,
LatexFormatOption,
MarkdownFormatOption,
OdpFormatOption,
OdsFormatOption,
OdtFormatOption,
PatentUsptoFormatOption,
PdfFormatOption,
PowerpointFormatOption,
VideoFormatOption,
WordFormatOption,
XBRLFormatOption,
XMLDocLangFormatOption,
XMLJatsFormatOption,
Expand Down Expand Up @@ -361,7 +406,7 @@ class DoclingPipelineConfig(BasePipelineConfig):
pipeline: ClassVar[PipelineType] = Field(frozen=True, default=PipelineType.DOCLING)

format_options: dict[InputFormat, DoclingFormatOption] = Field(
default_factory=_default_format_opts
default_factory=default_format_opts
)

settings: DoclingSettings = Field(default_factory=DoclingSettings)
Expand Down
46 changes: 8 additions & 38 deletions extract-core/extract_core/miner_u.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,10 @@
from collections.abc import Callable
from copy import copy
from enum import StrEnum
from functools import cache
from typing import Any, ClassVar
from typing import ClassVar, Literal

from mineru.config import VlmConfig
from mineru.types import Tier
from pydantic import Field
from pydantic_extra_types.language_code import LanguageAlpha2

Expand All @@ -21,48 +22,17 @@ class MinerUBackend(StrEnum):

class MinerUConfig(BaseModel):
backend: MinerUBackend = MinerUBackend.PIPELINE
enable_formula_extraction: bool = True
enable_table_extraction: bool = True
# TODO: use enum or literal here
parse_method: str = "auto"

def as_parse_kwargs(self) -> dict[str, Any]:
kwargs = copy(self._get_default_kwargs())
kwargs["backend"] = self.backend
kwargs["parse_method"] = self.parse_method
kwargs["formula_enable"] = self.enable_formula_extraction
kwargs["table_enable"] = self.enable_table_extraction
return kwargs

@classmethod
@cache
def _get_default_kwargs(cls) -> dict[str, Any]:
from mineru.utils.enum_class import MakeMode # noqa: PLC0415

return {
"server_url": None,
# We don't dump md directly we process, we dump the middle json in order
# to be able to get page indexes
"parse_method": "auto",
"dump_md": False,
"dump_middle_json": True,
"f_draw_layout_bbox": False,
"f_draw_span_bbox": False,
"f_dump_model_output": False, # might be useful for debug though
"f_dump_orig_pdf": False,
"f_dump_content_list": False, # might be useful for debug though
"start_page_id": 0,
"f_make_md_mode": MakeMode.MM_MD,
"image_analysis": True,
"end_page_id": None,
"client_side_output_generation": False,
}
tier: Tier = "basic"
parse_mode: Literal["auto", "txt", "ocr"] = "auto"
image_analysis: bool = True
vlm_config: VlmConfig | None = None


class MinerUPipelineConfig(BasePipelineConfig): # noqa: F821
pipeline: ClassVar[PipelineType] = Field(frozen=True, default=PipelineType.MINER_U)

config: MinerUConfig = Field(frozen=True, default=MinerUConfig())
config: MinerUConfig = Field(frozen=True, default_factory=MinerUConfig)
language: LanguageAlpha2 = Field(frozen=True, default="en")

@classmethod
Expand Down
6 changes: 6 additions & 0 deletions extract-core/extract_core/objects.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,12 @@ def to_docling(self) -> AcceleratorDevice:
raise ValueError(f"unsupported device {self}")


class PipelineSize(StrEnum):
SMALL = "small"
MEDIUM = "medium"
LARGE = "large"


class SupportedExt(StrEnum):
ADOC = ".adoc"
AFP = ".afp"
Expand Down
Loading
Loading