- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
215 lines
6.0 KiB
Python
215 lines
6.0 KiB
Python
#!/usr/bin/env python3
|
|
"""Main CLI entrypoint for Multilingual NLP Entity Inherence Classifier (POC)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
from src import __version__
|
|
from src.classifier import InherenceClassifier
|
|
from src.models import ClassificationError, ECPSnapshot, ErrorCode
|
|
|
|
# Ensure UTF-8 output streams across all platforms
|
|
if hasattr(sys.stdout, "reconfigure"):
|
|
try:
|
|
sys.stdout.reconfigure(encoding="utf-8")
|
|
except Exception:
|
|
pass
|
|
if hasattr(sys.stderr, "reconfigure"):
|
|
try:
|
|
sys.stderr.reconfigure(encoding="utf-8")
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
prog="classify.py",
|
|
description="Multilingual NLP Entity Inherence Classifier (POC)",
|
|
)
|
|
parser.add_argument(
|
|
"--ecp",
|
|
type=str,
|
|
required=True,
|
|
help="Path to the ECP Snapshot JSON file",
|
|
)
|
|
parser.add_argument(
|
|
"--content",
|
|
type=str,
|
|
required=True,
|
|
help="Path to the Markdown content file",
|
|
)
|
|
parser.add_argument(
|
|
"--output",
|
|
"-o",
|
|
type=str,
|
|
default=None,
|
|
help="Path to write the output JSON (default: prints to stdout)",
|
|
)
|
|
parser.add_argument(
|
|
"--enable-embeddings",
|
|
action="store_true",
|
|
default=False,
|
|
help="Enable optional Tier 2 vector embeddings adapter (default: false)",
|
|
)
|
|
parser.add_argument(
|
|
"--enable-llm",
|
|
action="store_true",
|
|
default=False,
|
|
help="Enable optional Tier 3 LLM fallback adapter (default: false)",
|
|
)
|
|
parser.add_argument(
|
|
"--version",
|
|
"-v",
|
|
action="version",
|
|
version=f"%(prog)s {__version__}",
|
|
)
|
|
return parser.parse_args(argv)
|
|
|
|
|
|
def emit_error(
|
|
error_code: ErrorCode,
|
|
message: str,
|
|
details: dict | None = None,
|
|
output_path: str | None = None,
|
|
) -> int:
|
|
err = ClassificationError(
|
|
error_code=error_code,
|
|
message=message,
|
|
details=details or {},
|
|
)
|
|
err_json = err.to_json_str(indent=2)
|
|
|
|
if output_path:
|
|
try:
|
|
out_file = Path(output_path)
|
|
out_file.parent.mkdir(parents=True, exist_ok=True)
|
|
out_file.write_text(err_json, encoding="utf-8")
|
|
except Exception:
|
|
pass
|
|
|
|
sys.stderr.write(err_json + "\n")
|
|
return 1
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
try:
|
|
args = parse_args(argv)
|
|
except SystemExit as e:
|
|
return int(e.code)
|
|
|
|
ecp_path = Path(args.ecp)
|
|
content_path = Path(args.content)
|
|
|
|
# 1. Validate ECP file existence and readability
|
|
if not ecp_path.is_file():
|
|
return emit_error(
|
|
ErrorCode.INVALID_ECP_JSON,
|
|
f"ECP snapshot file not found: '{args.ecp}'",
|
|
{"path": str(args.ecp)},
|
|
output_path=args.output,
|
|
)
|
|
|
|
try:
|
|
ecp_raw = ecp_path.read_text(encoding="utf-8")
|
|
except Exception as e:
|
|
return emit_error(
|
|
ErrorCode.INVALID_ECP_JSON,
|
|
f"Failed to read ECP file: {e}",
|
|
{"path": str(args.ecp), "error": str(e)},
|
|
output_path=args.output,
|
|
)
|
|
|
|
try:
|
|
ecp = ECPSnapshot.from_json_str(ecp_raw)
|
|
except ValueError as e:
|
|
error_msg = str(e)
|
|
if "Missing required field" in error_msg:
|
|
return emit_error(
|
|
ErrorCode.MISSING_REQUIRED_FIELD,
|
|
error_msg,
|
|
{"path": str(args.ecp)},
|
|
output_path=args.output,
|
|
)
|
|
return emit_error(
|
|
ErrorCode.INVALID_ECP_JSON,
|
|
error_msg,
|
|
{"path": str(args.ecp)},
|
|
output_path=args.output,
|
|
)
|
|
|
|
# 2. Validate Content file existence and readability
|
|
if not content_path.is_file():
|
|
return emit_error(
|
|
ErrorCode.INVALID_MARKDOWN,
|
|
f"Content markdown file not found: '{args.content}'",
|
|
{"path": str(args.content)},
|
|
output_path=args.output,
|
|
)
|
|
|
|
try:
|
|
content_raw = content_path.read_text(encoding="utf-8")
|
|
except Exception as e:
|
|
return emit_error(
|
|
ErrorCode.INVALID_MARKDOWN,
|
|
f"Failed to read content file: {e}",
|
|
{"path": str(args.content), "error": str(e)},
|
|
output_path=args.output,
|
|
)
|
|
|
|
if not content_raw or len(content_raw.strip()) < 5:
|
|
return emit_error(
|
|
ErrorCode.EMPTY_CONTENT,
|
|
"Content file is empty or contains insufficient text (minimum 5 non-whitespace characters required)",
|
|
{"path": str(args.content), "length": len(content_raw.strip()) if content_raw else 0},
|
|
output_path=args.output,
|
|
)
|
|
|
|
# 3. Execute Classification
|
|
classifier = InherenceClassifier(
|
|
enable_embeddings=args.enable_embeddings,
|
|
enable_llm=args.enable_llm,
|
|
)
|
|
|
|
try:
|
|
result = classifier.classify(ecp, content_raw)
|
|
except ValueError as e:
|
|
return emit_error(
|
|
ErrorCode.INVALID_MARKDOWN,
|
|
str(e),
|
|
{"path": str(args.content)},
|
|
output_path=args.output,
|
|
)
|
|
except Exception as e:
|
|
return emit_error(
|
|
ErrorCode.INVALID_MARKDOWN,
|
|
f"Classification processing error: {e}",
|
|
{"error": str(e)},
|
|
output_path=args.output,
|
|
)
|
|
|
|
result_json = result.to_json_str(indent=2)
|
|
|
|
# 4. Output results
|
|
if args.output:
|
|
try:
|
|
out_file = Path(args.output)
|
|
out_file.parent.mkdir(parents=True, exist_ok=True)
|
|
out_file.write_text(result_json, encoding="utf-8")
|
|
except Exception as e:
|
|
return emit_error(
|
|
ErrorCode.INVALID_MARKDOWN,
|
|
f"Failed to write output file: {e}",
|
|
{"output_path": str(args.output), "error": str(e)},
|
|
)
|
|
else:
|
|
sys.stdout.write(result_json + "\n")
|
|
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|