from app.services.pairing.embeddings import cosine, embedding_text


def test_text_combines_title_and_description():
    text = embedding_text({"name": "Oxford Shirt",
                           "description": "A white cotton shirt.",
                           "category": "shirts", "attributes": []})
    assert "Oxford Shirt" in text
    assert "white cotton" in text


def test_text_includes_category_and_attributes():
    # The attributes Phase 1 extracted are the whole reason similarity improved;
    # leaving them out of the embedded text would waste them.
    text = embedding_text({"name": "Shirt", "description": None,
                           "category": "shirts",
                           "attributes": [{"key": "color", "value": "white"}]})
    assert "shirts" in text
    assert "white" in text


def test_text_is_stable_for_the_same_product():
    product = {"name": "Shirt", "description": "d", "category": "c",
               "attributes": [{"key": "b", "value": "2"},
                              {"key": "a", "value": "1"}]}
    assert embedding_text(product) == embedding_text(dict(product))


def test_text_does_not_depend_on_attribute_order():
    # Attribute order is not stable across runs; an order-sensitive text would
    # re-embed the whole catalog for nothing.
    one = embedding_text({"name": "S", "description": "", "category": "c",
                          "attributes": [{"key": "a", "value": "1"},
                                         {"key": "b", "value": "2"}]})
    two = embedding_text({"name": "S", "description": "", "category": "c",
                          "attributes": [{"key": "b", "value": "2"},
                                         {"key": "a", "value": "1"}]})
    assert one == two


def test_cosine_of_identical_vectors_is_one():
    assert cosine([1.0, 0.0], [1.0, 0.0]) == 1.0


def test_cosine_of_a_zero_vector_is_zero_not_a_crash():
    assert cosine([0.0, 0.0], [1.0, 0.0]) == 0.0
