| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374 |
- import logging
- import re
- from pathlib import Path
- import cv2
- import yaml
- from fastapi import HTTPException
- from markdown_it import MarkdownIt
- from mdit_py_plugins.anchors import anchors_plugin as section_header
- from pydantic import ValidationError
- from app.schema import (
- DocumentCheckout,
- DocumentEmail,
- DocumentIndex,
- DocumentOrders,
- DocumentPage,
- DocumentProduct,
- DocumentProductIndex,
- DocumentSite,
- )
- logger = logging.getLogger(__name__)
- def md(text: str) -> str:
- """Helper function to convert parsed markdown-formatted text to HTML."""
- md = (
- MarkdownIt("commonmark", {"breaks": True, "html": True})
- .enable("table")
- .use(section_header)
- )
- return md.render(text)
- def get_files_by_extension(folder: Path, extensions: list[str], ignore_list: list = [], relative_path: bool = False) -> list[Path]:
- """Getting list of files specified by extensions"""
- file_list = []
- for extension in extensions:
- file_list.extend([f for f in list(folder.rglob(f'{extension}')) if f not in ignore_list])
- if relative_path:
- for i in range(len(file_list)):
- file_list[i] = file_list[i].relative_to(folder)
- return file_list
- def set_default_product_meta_fields(block) -> dict[str, str]:
- """
- Helper function to make sure all necessary fields expected by the
- document schema are present in dictionary. This prevents errors
- when in the text document some keys have been omitted (because
- unnecessary / empty), but are needed by the document scheme.
- """
- if block['template'] == 'product':
-
- if 'code' not in block:
- block['code'] = ''
- elif 'product_id' not in block:
- block['product_id'] = ''
- elif 'price' not in block:
- block['price'] = 0
- elif 'weight' not in block:
- block['weight'] = 0
- elif 'customs_tariff_code' not in block:
- block['customs_tariff_code'] = ''
- elif 'origin_country_iso' not in block:
- block['origin_country_iso'] = ''
- elif 'category' not in block:
- block['category'] = ''
- if 'options' in block and len(block['options']) > 0:
- if 'size' not in block['options']:
- block['options']['size'] = None
- if 'style' not in block['options']:
- block['options']['style'] = None
- else:
- block['options'] = {
- 'size':None,
- 'style': None,
- }
- if 'inventory' in block and len(block['inventory']) > 0:
- for item in block['inventory']:
- if 'size' not in item:
- item['size'] = None
- if 'style' not in item:
- item['style'] = None
- if 'amount' not in item:
- item['amount'] = 0
- else:
- block['inventory'] = [
- {
- 'size': None, 'style': None, 'amount': 0
- }
- ]
-
- return block
- def make_block_md(block: str) -> dict[str]:
- """Helper function to prepare a dictionary object for a block of type
- text. Block texts are free-form strings and not formatted with
- YAML, so we prepare the dictionary here.
- """
- block.strip()
- if block != "":
- return {"value": block, "type": "text"}
- def make_block_image(
- repo_path: str, meta_path: str, block: dict[str, str]
- ) -> dict[str, int]:
- """Helper function to enrich the image block with a few more info
- about the file it points to. We make use of this when displaying
- the image in the frontend template and prepare the correct srcset
- list of sizes we can display for the given image.
- """
- filepath = None
- if (meta_path != '.'):
- filepath = f"{repo_path}/{meta_path}/{block['url']}"
- else:
- filepath = f"{repo_path}/{block['url']}"
- img = cv2.imread(filepath)
- if img is not None and img.any():
- if 'caption' not in block:
- block['caption'] = ""
- block["info"] = {
- "width": round(img.shape[1]),
- "height": round(img.shape[0]),
- }
- return block
- else:
- return {
- "type": "image",
- "value": "",
- "url": "",
- "caption": "",
- "info": {
- "width": 0,
- "height": 0
- },
- }
- def parse_file(
- repo_path: str, input_path: str | Path, block_types: list[str]
- ) -> (
- DocumentSite
- | DocumentIndex
- | DocumentPage
- | DocumentProductIndex
- | DocumentProduct
- | DocumentCheckout
- | DocumentOrders
- | DocumentEmail
- ):
- """Parse file at given filepath: return a dictionary with meta
- dictionary and list of blocks.
- """
- document = {
- "meta": {
- "type": "",
- "title": "",
- "template": "",
- "path": "",
- },
- "blocks": [],
- }
- try:
- p = Path(input_path)
- except KeyError:
- document["meta"]["path"] = ""
- data = input_path.read_text()
- block_marker = "\n---\n"
- blocks = data.split(block_marker)
- blocks = [re.sub(r"^---\n", "", block) for block in blocks]
- # set meta_path to default value
- meta_path = document["meta"]["path"]
- for block in blocks:
- # we check if current block is a potential
- # yaml block, then check if any other block
- # has a field type in the ones "allowed".
- # in case the block.type has no matches
- # or it fails the yaml parser, we take it
- # as a markdown block
- try:
- block_yaml = yaml.safe_load(block)
- if block_yaml is not None:
- if type(block_yaml) is dict:
- # check if "valid" YAML dict is actually
- # a long string with a colon in the middle;
- # somehow this trips up the YAML parser and
- # we need to take care of it ourselves
- for k in block_yaml.keys():
- if len(k.split()) > 1 and ":" in k:
- block_txt = make_block_md(block)
- document["blocks"].append(block_txt)
- if "type" in block_yaml:
- if block_yaml["type"] == "meta":
- document["meta"] = set_default_product_meta_fields(block_yaml)
- for k, v in block_yaml.items():
- document["meta"][k] = v
- # set meta-path
- if "template" in block_yaml:
- doc_path = Path(p).relative_to(repo_path).parent
- document["meta"]["path"] = str(doc_path)
- meta_path = document["meta"]["path"]
- elif (
- "type" in block_yaml
- and block_yaml["type"].lower() in block_types
- ):
- if block_yaml["type"] == "image":
- make_block_image(repo_path, meta_path, block_yaml)
- document["blocks"].append(block_yaml)
- else:
- block_md = make_block_md(block)
- document["blocks"].append(block_md)
- except yaml.YAMLError:
- block_md = make_block_md(block)
- document["blocks"].append(block_md)
- logger.error("=== DEBUG PARSED DOCUMENT ===")
- logger.error(f"File being parsed: {input_path}")
- logger.error(f"Document meta keys: {list(document['meta'].keys())}")
- logger.error(f"Document meta values: {document['meta']}")
- logger.error(f"Template found: '{document['meta'].get('template', 'NOT FOUND')}'")
- logger.error(f"Number of content blocks: {len(document['blocks'])}")
- # Check if template is missing
- if not document['meta'].get('template'):
- logger.error("CRITICAL: No template found in document!")
- if blocks and len(blocks) > 0:
- logger.error(f"First block preview: {blocks[0][:500]}")
- try:
- if document["meta"]["template"] == "site":
- return DocumentSite(**document)
- elif document["meta"]["template"] == "index":
- return DocumentIndex(**document)
- elif document["meta"]["template"] == "page":
- return DocumentPage(**document)
- elif document["meta"]["template"] == "product-index":
- return DocumentProductIndex(**document)
- elif document["meta"]["template"] == "product":
- return DocumentProduct(**document)
- elif document["meta"]["template"] == "checkout":
- return DocumentCheckout(**document)
- elif document["meta"]["template"] == "orders":
- return DocumentOrders(**document)
- elif document["meta"]["template"] == "email":
- return DocumentEmail(**document)
- except ValidationError as e:
- logger.error("pydantic validation errors =>")
- for e in e.errors():
- logger.error(f"error => {e}")
- # Add more context about what failed
- logger.error(f"Validation failed for template: '{document['meta'].get('template', 'NO TEMPLATE')}'")
- logger.error(f"Document structure had keys: {list(document.keys())}")
- if 'meta' in document:
- logger.error(f"Meta had keys: {list(document['meta'].keys())}")
- raise FileNotFoundError
- def read_dir(
- repo_path: str,
- document_match: list[str],
- document_exclude: list[str],
- block_types: list[str],
- exclude_doc: str | None = None,
- tree: str | None = None,
- ):
- """Run get_files_by_extension with appropriate pattern matching to return list"""
- d = Path(repo_path).expanduser()
- pattern_matching = []
- [pattern_matching.append(f"*{ext}") for ext in document_match]
- ignore_list = []
- for file in document_exclude:
- ignore_list.append(Path(f"{repo_path}/{file}"))
- paths = get_files_by_extension(d, pattern_matching, ignore_list)
- paths = [p for p in paths if tree in str(p)]
- docs = []
- for p in paths:
- doc = parse_file(d, p, block_types)
- if doc.meta.path != exclude_doc:
- docs.append(doc)
- return docs
- def read_file(
- repo_path: str,
- filename: str,
- file_match: list[str],
- block_types: list[str],
- tree: str | None = None,
- ):
- """Helper function to read file from disk. Check if filepath's
- extension is allowed (by settings.toml) and filepath is tracked by
- git, then read its content.
- """
- filepath = None
- ext = file_match[0]
- if tree:
- p = f"{Path(repo_path)}/{tree}/{filename}"
- else:
- p = f"{Path(repo_path)}/{filename}"
- if Path(p).is_dir():
- file_path = f"{p}/index{ext}"
- if Path(file_path).is_file():
- filepath = file_path
- else:
- raise FileNotFoundError
- elif Path(f"{p}{ext}").is_file():
- filepath = f"{p}{ext}"
- if filepath:
- # get file path relative to the content repo
- # eg <path/to/repo/dir/file> => <dir>/<file>
- filename = str(Path(filepath).relative_to(repo_path))
- doc = parse_file(repo_path, Path(filepath), block_types)
- return doc
- else:
- raise HTTPException(status_code=404, detail="Nothing was found here.")
|