{"id":"https://openalex.org/W4321012295","doi":"https://doi.org/10.48550/arxiv.2302.06884","title":"Conservative State Value Estimation for Offline Reinforcement Learning","display_name":"Conservative State Value Estimation for Offline Reinforcement Learning","publication_year":2023,"publication_date":"2023-01-01","ids":{"openalex":"https://openalex.org/W4321012295","doi":"https://doi.org/10.48550/arxiv.2302.06884"},"language":"en","primary_location":{"is_oa":true,"landing_page_url":"https://arxiv.org/abs/2302.06884","pdf_url":null,"source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":["Cornell University"],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false},"type":"preprint","type_crossref":"posted-content","indexed_in":["arxiv","datacite"],"open_access":{"is_oa":true,"oa_status":"green","oa_url":"https://arxiv.org/abs/2302.06884","any_repository_has_fulltext":true},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5088627356","display_name":"Liting Chen","orcid":"https://orcid.org/0000-0002-7786-1226"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Chen, Liting","raw_affiliation_strings":[],"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5100716666","display_name":"Jie Yan","orcid":"https://orcid.org/0000-0003-4554-3132"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Yan, Jie","raw_affiliation_strings":[],"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5072366496","display_name":"Zhengdao Shao","orcid":null},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Shao, Zhengdao","raw_affiliation_strings":[],"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5083964854","display_name":"Lu Wang","orcid":"https://orcid.org/0000-0002-7305-1496"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Wang, Lu","raw_affiliation_strings":[],"affiliations":[]},{"author_position":"middle","author":{"id":"https://openalex.org/A5088646345","display_name":"Qingwei Lin","orcid":"https://orcid.org/0000-0003-2559-2383"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Lin, Qingwei","raw_affiliation_strings":[],"affiliations":[]},{"author_position":"last","author":{"id":"https://openalex.org/A5100331488","display_name":"Dongmei Zhang","orcid":"https://orcid.org/0000-0002-9230-2799"},"institutions":[],"countries":[],"is_corresponding":false,"raw_author_name":"Zhang, Dongmei","raw_affiliation_strings":[],"affiliations":[]}],"institution_assertions":[],"countries_distinct_count":0,"institutions_distinct_count":0,"corresponding_author_ids":[],"corresponding_institution_ids":[],"apc_list":null,"apc_paid":null,"fwci":null,"has_fulltext":false,"cited_by_count":2,"citation_normalized_percentile":{"value":0.778623,"is_in_top_1_percent":false,"is_in_top_10_percent":false},"cited_by_percentile_year":{"min":78,"max":84},"biblio":{"volume":null,"issue":null,"first_page":null,"last_page":null},"is_retracted":false,"is_paratext":false,"primary_topic":{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9931,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10462","display_name":"Reinforcement Learning in Robotics","score":0.9931,"subfield":{"id":"https://openalex.org/subfields/1702","display_name":"Artificial Intelligence"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10675","display_name":"Mechanical Circulatory Support Devices","score":0.9501,"subfield":{"id":"https://openalex.org/subfields/2204","display_name":"Biomedical Engineering"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10603","display_name":"Smart Grid Energy Management","score":0.9517,"subfield":{"id":"https://openalex.org/subfields/2208","display_name":"Electrical and Electronic Engineering"},"field":{"id":"https://openalex.org/fields/22","display_name":"Engineering"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/value","display_name":"Value (mathematics)","score":0.49952602},{"id":"https://openalex.org/keywords/penalty-method","display_name":"Penalty Method","score":0.4849782}],"concepts":[{"id":"https://openalex.org/C97541855","wikidata":"https://www.wikidata.org/wiki/Q830687","display_name":"Reinforcement learning","level":2,"score":0.8408926},{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.66510963},{"id":"https://openalex.org/C14646407","wikidata":"https://www.wikidata.org/wiki/Q1430750","display_name":"Bellman equation","level":2,"score":0.6430686},{"id":"https://openalex.org/C14036430","wikidata":"https://www.wikidata.org/wiki/Q3736076","display_name":"Function (biology)","level":2,"score":0.55210817},{"id":"https://openalex.org/C96250715","wikidata":"https://www.wikidata.org/wiki/Q965330","display_name":"Estimation","level":2,"score":0.53267086},{"id":"https://openalex.org/C48103436","wikidata":"https://www.wikidata.org/wiki/Q599031","display_name":"State (computer science)","level":2,"score":0.5275566},{"id":"https://openalex.org/C2776291640","wikidata":"https://www.wikidata.org/wiki/Q2912517","display_name":"Value (mathematics)","level":2,"score":0.49952602},{"id":"https://openalex.org/C6180225","wikidata":"https://www.wikidata.org/wiki/Q3411771","display_name":"Penalty method","level":2,"score":0.4849782},{"id":"https://openalex.org/C132459708","wikidata":"https://www.wikidata.org/wiki/Q744069","display_name":"Extrapolation","level":2,"score":0.47401863},{"id":"https://openalex.org/C126255220","wikidata":"https://www.wikidata.org/wiki/Q141495","display_name":"Mathematical optimization","level":1,"score":0.47400412},{"id":"https://openalex.org/C154945302","wikidata":"https://www.wikidata.org/wiki/Q11660","display_name":"Artificial intelligence","level":1,"score":0.44909012},{"id":"https://openalex.org/C119857082","wikidata":"https://www.wikidata.org/wiki/Q2539","display_name":"Machine learning","level":1,"score":0.41857252},{"id":"https://openalex.org/C61797465","wikidata":"https://www.wikidata.org/wiki/Q1188986","display_name":"Term (time)","level":2,"score":0.41454074},{"id":"https://openalex.org/C11413529","wikidata":"https://www.wikidata.org/wiki/Q8366","display_name":"Algorithm","level":1,"score":0.24506155},{"id":"https://openalex.org/C33923547","wikidata":"https://www.wikidata.org/wiki/Q395","display_name":"Mathematics","level":0,"score":0.18173656},{"id":"https://openalex.org/C105795698","wikidata":"https://www.wikidata.org/wiki/Q12483","display_name":"Statistics","level":1,"score":0.13213545},{"id":"https://openalex.org/C162324750","wikidata":"https://www.wikidata.org/wiki/Q8134","display_name":"Economics","level":0,"score":0.078483045},{"id":"https://openalex.org/C121332964","wikidata":"https://www.wikidata.org/wiki/Q413","display_name":"Physics","level":0,"score":0.0},{"id":"https://openalex.org/C62520636","wikidata":"https://www.wikidata.org/wiki/Q944","display_name":"Quantum mechanics","level":1,"score":0.0},{"id":"https://openalex.org/C86803240","wikidata":"https://www.wikidata.org/wiki/Q420","display_name":"Biology","level":0,"score":0.0},{"id":"https://openalex.org/C187736073","wikidata":"https://www.wikidata.org/wiki/Q2920921","display_name":"Management","level":1,"score":0.0},{"id":"https://openalex.org/C78458016","wikidata":"https://www.wikidata.org/wiki/Q840400","display_name":"Evolutionary biology","level":1,"score":0.0}],"mesh":[],"locations_count":3,"locations":[{"is_oa":true,"landing_page_url":"https://arxiv.org/abs/2302.06884","pdf_url":null,"source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":["Cornell University"],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false},{"is_oa":true,"landing_page_url":"http://arxiv.org/abs/2302.06884","pdf_url":"http://arxiv.org/pdf/2302.06884","source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":["Cornell University"],"type":"repository"},"license":null,"license_id":null,"version":"submittedVersion","is_accepted":false,"is_published":false},{"is_oa":false,"landing_page_url":"https://api.datacite.org/dois/10.48550/arxiv.2302.06884","pdf_url":null,"source":{"id":"https://openalex.org/S4393179698","display_name":"DataCite API","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I4210145204","host_organization_name":"DataCite","host_organization_lineage":["https://openalex.org/I4210145204"],"host_organization_lineage_names":["DataCite"],"type":"metadata"},"license":null,"license_id":null,"version":null}],"best_oa_location":{"is_oa":true,"landing_page_url":"https://arxiv.org/abs/2302.06884","pdf_url":null,"source":{"id":"https://openalex.org/S4306400194","display_name":"arXiv (Cornell University)","issn_l":null,"issn":null,"is_oa":true,"is_in_doaj":false,"is_core":false,"host_organization":"https://openalex.org/I205783295","host_organization_name":"Cornell University","host_organization_lineage":["https://openalex.org/I205783295"],"host_organization_lineage_names":["Cornell University"],"type":"repository"},"license":"other-oa","license_id":"https://openalex.org/licenses/other-oa","version":"submittedVersion","is_accepted":false,"is_published":false},"sustainable_development_goals":[{"display_name":"Peace, justice, and strong institutions","id":"https://metadata.un.org/sdg/16","score":0.81}],"grants":[],"datasets":[],"versions":[],"referenced_works_count":0,"referenced_works":[],"related_works":["https://openalex.org/W4361730764","https://openalex.org/W4296478327","https://openalex.org/W3099153698","https://openalex.org/W3038962357","https://openalex.org/W2386410636","https://openalex.org/W2042397106","https://openalex.org/W2025663273","https://openalex.org/W1968270095","https://openalex.org/W1965029248","https://openalex.org/W1960072520"],"abstract_inverted_index":{"Offline":[0],"reinforcement":[1],"learning":[2,25,170],"faces":[3],"a":[4,35,74,113],"significant":[5],"challenge":[6],"of":[7,158],"value":[8,41,97,124],"over-estimation":[9],"due":[10],"to":[11,24,33,38,48,89,147],"the":[12,16,19,44,119,122,131,134,137,149,167],"distributional":[13],"drift":[14],"between":[15],"dataset":[17],"and":[18,55,102,111,129,136,172],"current":[20],"learned":[21],"policy,":[22],"leading":[23],"failure":[26],"in":[27,43,117,153],"practice.":[28],"The":[29],"common":[30],"approach":[31,76],"is":[32,173],"incorporate":[34],"penalty":[36,84],"term":[37],"reward":[39],"or":[40],"estimation":[42,98,125],"Bellman":[45],"iterations.":[46],"Meanwhile,":[47],"avoid":[49],"extrapolation":[50],"on":[51,60,85],"out-of-distribution":[52],"(OOD)":[53],"states":[54,132],"actions,":[56],"existing":[57],"methods":[58,171],"focus":[59],"conservative":[61,79,100,123,168],"Q-function":[62,169],"estimation.":[63],"In":[64],"this":[65],"paper,":[66],"we":[67,108],"propose":[68],"Conservative":[69],"State":[70],"Value":[71],"Estimation":[72],"(CSVE),":[73],"new":[75],"that":[77,161],"learns":[78],"V-function":[80],"via":[81],"directly":[82],"imposing":[83],"OOD":[86],"states.":[87],"Compared":[88],"prior":[90],"work,":[91],"CSVE":[92,110],"allows":[93],"more":[94],"effective":[95],"state":[96,145],"with":[99,144],"guarantees":[101],"further":[103],"better":[104,165],"policy":[105],"optimization.":[106],"Further,":[107],"apply":[109],"develop":[112],"practical":[114],"actor-critic":[115],"algorithm":[116],"which":[118],"critic":[120],"does":[121],"by":[126],"additionally":[127],"sampling":[128],"penalizing":[130],"\\emph{around}":[133],"dataset,":[135],"actor":[138],"applies":[139],"advantage":[140],"weighted":[141],"updates":[142],"extended":[143],"exploration":[146],"improve":[148],"policy.":[150],"We":[151],"evaluate":[152],"classic":[154],"continual":[155],"control":[156],"tasks":[157],"D4RL,":[159],"showing":[160],"our":[162],"method":[163],"performs":[164],"than":[166],"strongly":[174],"competitive":[175],"among":[176],"recent":[177],"SOTA":[178],"methods.":[179]},"cited_by_api_url":"https://api.openalex.org/works?filter=cites:W4321012295","counts_by_year":[{"year":2024,"cited_by_count":2}],"updated_date":"2025-01-01T23:31:30.496645","created_date":"2023-02-17"}