【问题标题】:How do you make dataframe match datetime dates?你如何使数据框匹配日期时间日期?
【发布时间】:2021-08-16 04:31:35
【问题描述】:

所以我正在为 Tkinter GUI 编写代码,在其中,代码从 FRED 中提取数据并使用它来呈现图形。一开始有一个选项可以将提取的数据保存在 CSV 文件中,这样您就可以在没有互联网的情况下运行它。但是当代码运行以使用 CSV 时,比例会发生变化,它给了我一个类似this 的图表。我认为这与没有记住日期时间数据有关。当前代码情况如下:

导入:from tkinter import *, from tkinter import ttk, pandas_datareader as pdr, pandas as pd, from datetime import datetime

数据调用示例:

def getBudgetData():
    '''
    PURPOSE: Get the government budget balance data
    INPUTS: None
    OUTPUTS: The dataframe of the selected country
    '''
    
    global namedCountry

    # Reads what country is in the combobox when selected, then gives the index value so the correct
    # code is used in the graph
    namedCountry = countryCombo.get()
    selectedCountry = countryOptions.index(namedCountry)
    df = dfBudget[dfBudget.columns[selectedCountry]]

    return df

获取/读取数据帧的代码

def readDataframeCSV():
    global dfCPIQuarterly, dfCPIMonthly, dfGDP, dfUnemployment, dfCashRate, dfBudget
    dfCPIQuarterly = pd.read_csv('dataframes\dfCPIQuarterly.csv', infer_datetime_format = True)  
    dfCPIMonthly = pd.read_csv('dataframes\dfCPIMonthly.csv')
    dfGDP = pd.read_csv('dataframes\dfGDP.csv')
    dfUnemployment = pd.read_csv('dataframes\dfUnemployment.csv')
    dfCashRate = pd.read_csv('dataframes\dfCashRate.csv')
    dfBudget = pd.read_csv('dataframes\dfBudget.csv')

def LogDiff(x, frequency):
    '''
    PURPOSE: Transform level data into growth
    INPUTS: x (time series), frequency (frequency of time series)
    OUTPUTS: x_diff (growth rate of time series)
    REFERENCE: Tau, Ran, & Chris Brookes. (2019). Python Guide to accompany
               introductary econometrics for finance (4th Edition). 
               Cambridge University Press.
    '''
    
    x_diff = 100*log(x/x.shift(frequency))
    x_diff = x_diff.dropna()
    return x_diff

def getAllFredData():
    '''
    PURPOSE: Extract all required data from FRED
    INPUTS: None
    OUTPUTS: Dataframes of all time series
    REFERENCE: https://fred.stlouisfed.org/
    '''
    
    global dfCPIQuarterly, dfCPIMonthly, dfGDP, dfUnemployment, dfCashRate, dfBudget
    
    # Country codes
    countryCPIQuarterlyCodes = ['AUSCPIALLQINMEI', 'NZLCPIALLQINMEI']
    countryCPIMonthlyCodes = ['CPALCY01CAM661N', 'JPNCPIALLMINMEI', 'GBRCPIALLMINMEI', 'CPIAUCSL']
    countryGDPCodes = ['AUSGDPRQDSMEI', 'NAEXKP01CAQ189S', 'JPNRGDPEXP',
                    'NAEXKP01NZQ189S', 'CLVMNACSCAB1GQUK', 'GDPC1']
    countryUnemploymentCodes = ['LRUNTTTTAUQ156S', 'LRUNTTTTCAQ156S', 'LRUN64TTJPQ156S',
                    'LRUNTTTTNZQ156S', 'LRUNTTTTGBQ156S', 'LRUN64TTUSQ156S']
    countryCashRateCodes = ['IR3TBB01AUM156N', 'IR3TIB01CAM156N', 'INTDSRJPM193N',
                    'IR3TBB01NZM156N', 'IR3TIB01GBM156N', 'FEDFUNDS']
    countryBudgetCodes = ['GGNLBAAUA188N', 'GGNLBACAA188N', 'GGNLBAJPA188N',
                    'NZLGGXCNLG01GDPPT', 'GGNLBAGBA188N', 'FYFSGDA188S']
    
    # Inflation
    dfCPIQuarterly = pdr.DataReader(countryCPIQuarterlyCodes,
                                        'fred', start, end)
        
    for country in countryCPIQuarterlyCodes:
        dfCPIQuarterly[country] = pd.DataFrame({"Inflation rate":LogDiff(dfCPIQuarterly[country], 4)})
            
            
    dfCPIMonthly = pdr.DataReader(countryCPIMonthlyCodes,
                                               'fred', start, end)
                
    for country in countryCPIMonthlyCodes:
        dfCPIMonthly[country] = pd.DataFrame({"Inflation rate":LogDiff(dfCPIMonthly[country], 12)})
    
        
    # GDP
    dfGDP = pdr.DataReader(countryGDPCodes, 
                           'fred', start, end)
         
    for country in countryGDPCodes:
        dfGDP[country] = pd.DataFrame({"Economic Growth":LogDiff(dfGDP[country], 4)})
            
    # Unemployment
    dfUnemployment = pdr.DataReader(countryUnemploymentCodes,
                                    'fred', start, end)
    
    # Cash Rate
    dfCashRate = pdr.DataReader(countryCashRateCodes,
                                 'fred', start, end)

    # Budget
    dfBudget = pdr.DataReader(countryBudgetCodes,
                              'fred', start, end)
    
    print('')
    saveToCSVLoop = True
    while saveToCSVLoop == True:
        saveToCSV = input('Would you like to save the dataframes to a CSV file so start-up will be quicker next (y or n): ')
        if saveToCSV == 'y':
            dfCPIQuarterly.to_csv('dataframes\dfCPIQuarterly.csv', index = True)
            dfCPIMonthly.to_csv('dataframes\dfCPIMonthly.csv', index = False)
            dfGDP.to_csv('dataframes\dfGDP.csv', index = False)
            dfUnemployment.to_csv('dataframes\dfUnemployment.csv', index = False)
            dfCashRate.to_csv('dataframes\dfCashRate.csv', index = False)
            dfBudget.to_csv('dataframes\dfBudget.csv', index = False)
            saveToCSVLoop = False
        elif saveToCSV == 'n':
            saveToCSVLoop = False
        else:
            print('\nNot a valid option')
            sleep(1)

【问题讨论】:

    标签: python python-3.x pandas datetime


    【解决方案1】:

    如果没有 csv 数据,很难为您提供帮助。可能是日期没有正确保存,或者没有正确解释。也许您可以先尝试解析日期时间。看起来好像没有年份,或者预期的年份实际上是一个月?

    因为它从 1970 年开始,我有一种感觉,它将您的时间解释为 unix 纪元,而不是正常的 yyyymmdd 类型的日期。尝试打印dfCPIQuarterly,看看它是否看起来像一个日期。从 csv 读取它时,也许您不应该使用 infer_datetime_format = True,但如果没有更多细节,就很难判断。

    【讨论】:

    猜你喜欢
    • 1970-01-01
    • 1970-01-01
    • 2021-12-21
    • 2016-06-18
    • 2021-12-27
    • 1970-01-01
    • 1970-01-01
    • 1970-01-01
    • 1970-01-01
    相关资源
    最近更新 更多