
时间:2018-10-23 14:05:24

标签: python xml beautifulsoup lxml


    <?xml version="1.0" encoding="UTF-8" ?>
    <genre category="Action">
        <decade years="1980s">
            <movie favorite="True" title="Indiana Jones: The raiders of the lost Ark">
                <format multiple="No">DVD</format>
                'Archaeologist and adventurer Indiana Jones 
                is hired by the U.S. government to find the Ark of the 
                Covenant before the Nazis.'
               <movie favorite="True" title="THE KARATE KID">
               <format multiple="Yes">DVD,Online</format>
               <description>None provided.</description>
            <movie favorite="False" title="Back 2 the Future">
               <format multiple="False">Blu-ray</format>
               <description>Marty McFly</description>
        <decade years="1990s">
            <movie favorite="False" title="X-Men">
               <format multiple="Yes">dvd, digital</format>
               <description>Two mutants come to a private academy for their kind whose resident superhero team must 
               oppose a terrorist organization with similar powers.</description>
            <movie favorite="True" title="Batman Returns">
               <format multiple="No">VHS</format>
               <movie favorite="False" title="Reservoir Dogs">
               <format multiple="No">Online</format>
               <description>WhAtEvER I Want!!!?!</description>

    <genre category="Thriller">
        <decade years="1970s">
            <movie favorite="False" title="ALIEN">
                <format multiple="Yes">DVD</format>
        <decade years="1980s">
            <movie favorite="True" title="Ferris Bueller's Day Off">
                <format multiple="No">DVD</format>
                <description>Funny movie about a funny guy</description>
            <movie favorite="FALSE" title="American Psycho">
                <format multiple="No">blue-ray</format>
                <description>psychopathic Bateman</description>


First output  {'Action':['Indiana Jones: The raiders of the lost Ark', 'THE KARATE KID', 'Back 2 the Future','X-Men', 'Batman Returns', 'Reservoir Dogs']}
second output  {'movies':'description'}
third output   {'movies': 'year'}


from lxml import etree
parser = etree.XMLParser()
tree= etree.parse('movies.xml', parser)
data= tree.find("genre[@category='Action']")
json= {}
for child in enumerate(data.getchildren()):
    temp = {}
    for content in child[1].getchildren():
        temp[content.attrib.get('title')] =  content.text.strip()
        json[child[0]] = temp.keys()

1 个答案:

答案 0 :(得分:4)


import json

from lxml import etree

XSL = '''<xsl:stylesheet version="1.0" xmlns:xsl="http://www.w3.org/1999/XSL/Transform" xmlns:fo="http://www.w3.org/1999/XSL/Format">
    <xsl:output method="text"/>

    <xsl:template match="/collection">

    <xsl:template match="genre">
            <xsl:value-of select="@category"/>
        <xsl:text>": [</xsl:text>
        <xsl:for-each select="descendant::movie" >
                <xsl:value-of select="@title"/>
            <xsl:if test="position() != last()">
                <xsl:text>, </xsl:text>
        <xsl:if test="following-sibling::*">

    <xsl:template match="text()"/>

# load input
dom = etree.parse('movies.xml')
# load XSLT
transform = etree.XSLT(etree.fromstring(XSL))

# apply XSLT on loaded dom
json_text = str(transform(dom))

# json_text contains the data converted to JSON format.
# you can use it with the JSON API. Example:
data = json.loads(json_text)


{'Action': ['Indiana Jones: The raiders of the lost Ark', 'THE KARATE KID', 'Back 2 the Future', 'X-Men', 'Batman Returns', 'Reservoir Dogs'], 'Thriller': ['ALIEN', "Ferris Bueller's Day Off", 'American Psycho']}
