Source code for image_suggestions.unillustratable

#!/usr/bin/env python3
# -*- coding: utf-8 -*-

"""A set of utility functions that identify irrelevant images
and unsuitable article or section candidates.
"""

import string

from pyspark.sql import Column, DataFrame, SparkSession
from pyspark.sql import functions as F

from image_suggestions import shared


#: ASCII punctuation characters to be stripped from section titles.
#: Include the ASCII white space, don't strip round brackets.
STRIP_CHARS = string.punctuation.replace('()', ' ')
#: All kinds of white space to be substituted for the ASCII one;
#: underscores turn into spaces as well.
SUBSTITUTE_PATTERN = r'[\s_]'

#: If an article's Wikidata `item <https://www.wikidata.org/wiki/WD:GLOSS#Item>`_ is an
#: `instance of <https://www.wikidata.org/wiki/Property:P31>`_ one of the items in this list,
#: then it's not suitable for getting suggestions.
UNILLUSTRATABLE_P31 = (
    'Q577',  # Year
    'Q29964144',  # Year BC
    'Q3186692',  # Calendar year
    'Q3311614',  # Century leap year
    'Q14795564',  # Recurrent timeframe
    'Q101352',  # Family name
    'Q82799',  # Name
    'Q21199',  # Natural number
    'Q28920044',  # Positive integer
    'Q28920052',  # Non-negative integer
    'Q13406463',  # List
    'Q4167410',  # Disambiguation page
    'Q22808320',  # Wikimedia human name disambiguation page
    'Q98645843',  # Wikimedia music-related list
    'Q17099416',  # Wikimedia list of songs
    'Q100775261',  # Wikimedia list of musical works
)

# We have a stored DataFrame containing images that are in placeholder categories.
# See `maintenance/generate_image_placeholders.py`.
PLACEHOLDER_IMAGE_PARQUET = 'image_placeholders'

#: Image file names containing these substrings are probably icons or placeholders.
PLACEHOLDER_IMAGE_SUBSTRINGS = (
    'flag',
    'noantimage',
    'no_free_image',
    'image_manquante',
    'replace_this_image',
    'disambig',
    'regions',
    'map',
    'default',
    'defaut',
    'falta_imagem_',
    'imageNA',
    'noimage',
    'noenzyimage',
)

PLACEHOLDER_IMAGE_ALLOWED_FILE_EXTENSIONS = (
    '.bmp',
    '.jpeg',
    '.jpg',
    '.png',
    '.tif',
    '.tiff',
)


[docs] def get_disallowed_substrings_regex( substrings: tuple[str, ...] = PLACEHOLDER_IMAGE_SUBSTRINGS, ) -> str: """Build a regular expression to detect image file names that may be icons or placeholders. :param substrings: a tuple of substrings that indicate icons or placeholders in image file names :return: the regular expression that matches the given substrings """ return '(?i)({})'.format('|'.join(substrings))
[docs] def get_allowed_suffixes_regex( suffixes: tuple[str, ...] = PLACEHOLDER_IMAGE_ALLOWED_FILE_EXTENSIONS, ) -> str: """Build a regular expression to detect image file extensions that typically hold valid images. :param suffixes: a tuple of suffixes that indicate valid image file extensions :return: the regular expression that matches the given suffixes """ return '(?i)({})$'.format('|'.join(suffixes))
[docs] def read_section_images_parquet( spark: SparkSession, section_images_parquet: str ) -> DataFrame: # pragma: no cover """Load images available in all sections of all articles of all Wikipedias, as output by :mod:`imagerec.article_images`. :param spark: an active Spark session :param section_images_parquet: a HDFS path to a parquet generated by :mod:`imagerec.article_images` :return: the dataframe of: - item_id (string) - page Wikidata `QID <https://www.wikidata.org/wiki/Wikidata:Glossary#QID>`_ - page_id (string) - page ID - page_title (string) - page title, in original case and underscored - article_images (array<struct<heading:string,images:array<string>>>) - images per section per page - wiki_db (string) - wiki project """ return spark.read.parquet(section_images_parquet)
[docs] def get_section_images( spark: SparkSession, section_images_parquet: str ) -> DataFrame: # pragma: no cover """Explode a dataframe as loaded by :func:`read_section_images_parquet` for easier processing. :param spark: an active Spark session :param section_images_parquet: a HDFS path to a parquet generated by :mod:`imagerec.article_images` :return: the dataframe of: - wiki_db (string) - wiki project - page_id (string) - page ID - page_title (string) - page title, in original case and underscored - section_heading (string) - page section, in URL anchor format. More details in :func:`section_topics.pipeline.wikitext_headings_to_anchors` - image (string) - Commons image file name """ article_images = read_section_images_parquet(spark, section_images_parquet) return ( article_images.select('*', F.explode('article_images').alias('ais_array')) .select('*', F.explode('ais_array.images').alias('image')) .where(F.col('image').isNotNull()) .select( 'wiki_db', 'page_id', 'page_title', F.col('ais_array.heading').alias('section_heading'), 'image', ) .orderBy('page_title', 'section_heading', 'image') )
[docs] def get_non_illustratable_item_ids( spark: SparkSession, hive_db: str, weekly_snapshot: str ) -> DataFrame: """Gather Wikidata QIDs that aren't suitable for getting suggestions. See :const:`UNILLUSTRATABLE_P31`. :param spark: an active Spark session :param hive_db: a Data Lake's `Hive <https://hive.apache.org/>`_ database name :param weekly_snapshot: a ``YYYY-MM-DD`` date :return: the dataframe of Wikidata QIDs """ wikidata_items_with_p31 = shared.load_wikidata_items_with_p31( spark, hive_db, weekly_snapshot ) return wikidata_items_with_p31.where( F.col('value').isin(list(UNILLUSTRATABLE_P31)) ).select('item_id')
[docs] def read_denylist_parquet( spark: SparkSession, denylist_parquet: str ) -> DataFrame: # pragma: no cover """Load denylisted section titles. :param spark: an active Spark session :param denylist_parquet: a HDFS path to a parquet generated by :mod:`section_topics.scripts.gather_section_titles_denylist` :return: the dataframe of: - wiki_db (string) - wiki project - section_heading (string) - page section, in URL anchor format. """ return spark.read.parquet(denylist_parquet)
[docs] def get_non_illustratable_sections( spark: SparkSession, denylist_parquet: str, dataframe: DataFrame, wiki_column: Column, heading_column: Column, ) -> DataFrame: """Gather all Wikipedia article section headings that aren't suitable for getting suggestions. :param spark: an active Spark session :param denylist_parquet: a HDFS path to a parquet generated by :mod:`section_topics.scripts.gather_section_titles_denylist` :param dataframe: a dataframe of irrelevant section headings :param wiki_column: a ``dataframe``'s column of wikis :param heading_column: a ``dataframe``'s column of section headings :return: the dataframe of: - wiki_db (string) - wiki project - section_heading (string) - page section, in URL anchor format. More details in :func:`section_topics.pipeline.wikitext_headings_to_anchors` """ denylist_dataframe = read_denylist_parquet(spark, denylist_parquet) return denylist_dataframe.join( dataframe, on=[ denylist_dataframe.wiki_db == wiki_column, normalize_heading_column(denylist_dataframe.section_heading) == normalize_heading_column(heading_column), ], how='inner', ).select( wiki_column, heading_column.alias('section_heading'), )
[docs] def get_images_in_placeholder_categories( spark: SparkSession, ) -> DataFrame: # pragma: no cover """Load images that belong to the placeholder Commons category. :param spark: an active Spark session :return: the dataframe of: - cl_from (bigint) - Commons page ID - cl_to (string) - Commons category page title, in original case and underscored - cl_type (string) - ``'file'`` - page_title (string) - Commons page title, in original case and underscored """ return spark.read.parquet(PLACEHOLDER_IMAGE_PARQUET)
# Copied from https://gitlab.wikimedia.org/repos/structured-data/section-topics/-/blob/main/section_topics/pipeline.py
[docs] def normalize_heading_column( column: Column, substitute_pattern: str = SUBSTITUTE_PATTERN, strip_chars: str = STRIP_CHARS, ) -> Column: """Same as :func:`section_topics.pipeline.normalize_heading_column`.""" # These chars will be surrounded by `[]` to form a regexp group, so escape literal `[` and `]`. # Also double the backslash to prevent it from being an escape char itself. escaped_strip_chars = ( strip_chars.replace('\\', '\\\\').replace('[', r'\[').replace(']', r'\]') ) return F.lower( F.regexp_replace( F.regexp_replace( F.regexp_replace( column, '_[0-9]+$', '', ), substitute_pattern, ' ', ), f'^[{escaped_strip_chars}]*(.*?)[{escaped_strip_chars}]*$', '$1', ) )